{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:55.970121Z","iopub.execute_input":"2021-07-03T04:11:55.970483Z","iopub.status.idle":"2021-07-03T04:11:55.984932Z","shell.execute_reply.started":"2021-07-03T04:11:55.970442Z","shell.execute_reply":"2021-07-03T04:11:55.984008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# ====================================================\n# Library\n# ====================================================\nimport sys\nsys.path.append('../input/d/kozodoi/timm-pytorch-image-models/pytorch-image-models-master')\n\nimport os\nimport math\nimport time\nimport random\nimport shutil\nfrom pathlib import Path\nfrom contextlib import contextmanager\nfrom collections import defaultdict, Counter\n\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn import preprocessing\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\n\nfrom tqdm.auto import tqdm\nfrom functools import partial\n\nimport cv2\nfrom PIL import Image\n\nfrom matplotlib import pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.optim import Adam, SGD\nimport torchvision.models as models\nfrom torch.nn.parameter import Parameter\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.optim.lr_scheduler import CosineAnnealingWarmRestarts, CosineAnnealingLR, ReduceLROnPlateau\n\nfrom albumentations import (\n    Compose, OneOf, Normalize, Resize, RandomResizedCrop, RandomCrop, HorizontalFlip, VerticalFlip, \n    RandomBrightness, RandomContrast, RandomBrightnessContrast, Rotate, ShiftScaleRotate, Cutout, \n    IAAAdditiveGaussianNoise, Transpose\n    )\nfrom albumentations.pytorch import ToTensorV2\nfrom albumentations import ImageOnlyTransform\n\nimport timm\n\nfrom torch.cuda.amp import autocast, GradScaler\n\nimport warnings \nwarnings.filterwarnings('ignore')\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nMODEL_DIR = '../input/study-level-training/'\nOUTPUT_DIR = './'\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:55.986803Z","iopub.execute_input":"2021-07-03T04:11:55.987195Z","iopub.status.idle":"2021-07-03T04:11:56.006574Z","shell.execute_reply.started":"2021-07-03T04:11:55.987153Z","shell.execute_reply":"2021-07-03T04:11:56.005722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_DIR2 = '../input/none-0-1-binary-training/'","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:56.008242Z","iopub.execute_input":"2021-07-03T04:11:56.008642Z","iopub.status.idle":"2021-07-03T04:11:56.015374Z","shell.execute_reply.started":"2021-07-03T04:11:56.008593Z","shell.execute_reply":"2021-07-03T04:11:56.01455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-03T04:11:56.016864Z","iopub.execute_input":"2021-07-03T04:11:56.017212Z","iopub.status.idle":"2021-07-03T04:11:56.024406Z","shell.execute_reply.started":"2021-07-03T04:11:56.017169Z","shell.execute_reply":"2021-07-03T04:11:56.023458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n# df = df.head(200)\n\n\nif df.shape[0] == 2477:\n    fast_sub = True\n    fast_df = pd.DataFrame(([['00086460a852_study', 'negative 1 0 0 1 1'], \n                         ['000c9c05fd14_study', 'negative 1 0 0 1 1'], \n                         ['65761e66de9f_image', 'none 1 0 0 1 1'], \n                         ['51759b5579bc_image', 'none 1 0 0 1 1']]), \n                       columns=['id', 'PredictionString'])\nelse:\n    fast_sub = False\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:56.025882Z","iopub.execute_input":"2021-07-03T04:11:56.026567Z","iopub.status.idle":"2021-07-03T04:11:56.041443Z","shell.execute_reply.started":"2021-07-03T04:11:56.026529Z","shell.execute_reply":"2021-07-03T04:11:56.040624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# .dcm to .png","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:56.164055Z","iopub.execute_input":"2021-07-03T04:11:56.164549Z","iopub.status.idle":"2021-07-03T04:11:56.415023Z","shell.execute_reply.started":"2021-07-03T04:11:56.164504Z","shell.execute_reply":"2021-07-03T04:11:56.414149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:56.416327Z","iopub.execute_input":"2021-07-03T04:11:56.416687Z","iopub.status.idle":"2021-07-03T04:11:56.422805Z","shell.execute_reply.started":"2021-07-03T04:11:56.416647Z","shell.execute_reply":"2021-07-03T04:11:56.420926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsplit = 'test'\nsave_dir = f'/kaggle/tmp/{split}/'\n\nos.makedirs(save_dir, exist_ok=True)\n\nsave_dir = f'/kaggle/tmp/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    study = '00086460a852' + '_study.png'\n    im.save(os.path.join(save_dir, study))\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    study = '000c9c05fd14' + '_study.png'\n    im.save(os.path.join(save_dir, study))\nelse:   \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=512)  \n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:56.424308Z","iopub.execute_input":"2021-07-03T04:11:56.424841Z","iopub.status.idle":"2021-07-03T04:11:57.910742Z","shell.execute_reply.started":"2021-07-03T04:11:56.424775Z","shell.execute_reply":"2021-07-03T04:11:57.909675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\nsave_dir = f'/kaggle/tmp/{split}/image/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=512)  \n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)\nmeta = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:57.912013Z","iopub.execute_input":"2021-07-03T04:11:57.91238Z","iopub.status.idle":"2021-07-03T04:11:58.290695Z","shell.execute_reply.started":"2021-07-03T04:11:57.912341Z","shell.execute_reply":"2021-07-03T04:11:58.289833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study predict","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nif fast_sub:\n    df = fast_df.copy()\nelse:\n    df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n    \n# df = df.head(200)\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\n\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.292033Z","iopub.execute_input":"2021-07-03T04:11:58.292387Z","iopub.status.idle":"2021-07-03T04:11:58.300936Z","shell.execute_reply.started":"2021-07-03T04:11:58.292348Z","shell.execute_reply":"2021-07-03T04:11:58.299914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\nif fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n# sub_df = sub_df.head(200)\nsub_df = sub_df[:study_len]\ntest_paths = f'/kaggle/tmp/{split}/study/' \n\nsub_df['negative'] = 0\nsub_df['typical'] = 0\nsub_df['indeterminate'] = 0\nsub_df['atypical'] = 0\n\n\nlabel_cols = sub_df.columns[2:]","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.302471Z","iopub.execute_input":"2021-07-03T04:11:58.302836Z","iopub.status.idle":"2021-07-03T04:11:58.312689Z","shell.execute_reply.started":"2021-07-03T04:11:58.302785Z","shell.execute_reply":"2021-07-03T04:11:58.311879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.314101Z","iopub.execute_input":"2021-07-03T04:11:58.314526Z","iopub.status.idle":"2021-07-03T04:11:58.329806Z","shell.execute_reply.started":"2021-07-03T04:11:58.314447Z","shell.execute_reply":"2021-07-03T04:11:58.328844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.331214Z","iopub.execute_input":"2021-07-03T04:11:58.331744Z","iopub.status.idle":"2021-07-03T04:11:58.337538Z","shell.execute_reply.started":"2021-07-03T04:11:58.331697Z","shell.execute_reply":"2021-07-03T04:11:58.336694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.339091Z","iopub.execute_input":"2021-07-03T04:11:58.339558Z","iopub.status.idle":"2021-07-03T04:11:58.349849Z","shell.execute_reply.started":"2021-07-03T04:11:58.339521Z","shell.execute_reply":"2021-07-03T04:11:58.348996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug=False\n    num_workers=4\n    model_name='tf_efficientnet_b7_ns'\n    model_name2='tf_efficientnet_b3'\n\n\n    size=512\n    batch_size=64\n    seed=42\n    target_size=4\n    target_cols=['Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']\n    n_fold=5\n    trn_fold=[0, 1, 2, 3,4]","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.354833Z","iopub.execute_input":"2021-07-03T04:11:58.35514Z","iopub.status.idle":"2021-07-03T04:11:58.359838Z","shell.execute_reply.started":"2021-07-03T04:11:58.355114Z","shell.execute_reply":"2021-07-03T04:11:58.358773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n# Utils\n# ====================================================\ndef get_score(y_true, y_pred):\n    scores = []\n    for i in range(y_true.shape[1]):\n        score = roc_auc_score(y_true[:,i], y_pred[:,i])\n        scores.append(score)\n    avg_score = np.mean(scores)\n    return avg_score, scores\n\n\ndef get_result(result_df):\n    preds = result_df[[f'pred_{c}' for c in CFG.target_cols]].values\n    labels = result_df[CFG.target_cols].values\n    score, scores = get_score(labels, preds)\n    LOGGER.info(f'Score: {score:<.4f}  Scores: {np.round(scores, decimals=4)}')\n\n\n@contextmanager\ndef timer(name):\n    t0 = time.time()\n    LOGGER.info(f'[{name}] start')\n    yield\n    LOGGER.info(f'[{name}] done in {time.time() - t0:.0f} s.')\n\n\ndef init_logger(log_file=OUTPUT_DIR+'inference.log'):\n    from logging import getLogger, INFO, FileHandler,  Formatter,  StreamHandler\n    logger = getLogger(__name__)\n    logger.setLevel(INFO)\n    handler1 = StreamHandler()\n    handler1.setFormatter(Formatter(\"%(message)s\"))\n    handler2 = FileHandler(filename=log_file)\n    handler2.setFormatter(Formatter(\"%(message)s\"))\n    logger.addHandler(handler1)\n    logger.addHandler(handler2)\n    return logger\n\nLOGGER = init_logger()\n\n\ndef seed_torch(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed_torch(seed=CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.362661Z","iopub.execute_input":"2021-07-03T04:11:58.363155Z","iopub.status.idle":"2021-07-03T04:11:58.378791Z","shell.execute_reply.started":"2021-07-03T04:11:58.363117Z","shell.execute_reply":"2021-07-03T04:11:58.377883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n#test = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/sample_submission.csv')\ntest = sub_df\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.380202Z","iopub.execute_input":"2021-07-03T04:11:58.380737Z","iopub.status.idle":"2021-07-03T04:11:58.39287Z","shell.execute_reply.started":"2021-07-03T04:11:58.380671Z","shell.execute_reply":"2021-07-03T04:11:58.391873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# file_path = glob.glob(f'../input/siim-covid19-detection/test/' + {} +'/*/*')[0] \n# file_path","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.394346Z","iopub.execute_input":"2021-07-03T04:11:58.394907Z","iopub.status.idle":"2021-07-03T04:11:58.400026Z","shell.execute_reply.started":"2021-07-03T04:11:58.39487Z","shell.execute_reply":"2021-07-03T04:11:58.399026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, df,study, transform=None):\n        self.df = df\n        self.file_names = df['id'].values\n        self.transform = transform\n        self.study = study\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        \n        file_name = self.file_names[idx]\n        if self.study:\n        #file_path = f'{TEST_PATH}/{file_name}.jpg'\n            file_path = f'/kaggle/tmp/{split}/study/' + file_name +'.png'   \n        else:\n            file_path = f'/kaggle/tmp/{split}/image/' + file_name +'.png'  \n        #print(file_path.shape)\n\n\n        image = cv2.imread(file_path)\n        #print(image.shape)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        if self.transform:\n            augmented = self.transform(image=image)\n            image = augmented['image']\n        return image\n    \n    \n    \ndef get_transforms(*, data):\n    \n    if data == 'train':\n        return Compose([\n            Resize(CFG.size, CFG.size),\n\n            ToTensorV2(),\n        ])\n\n    elif data == 'valid':\n        return Compose([\n            Resize(CFG.size, CFG.size),\n\n            ToTensorV2(),\n        ])","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.401339Z","iopub.execute_input":"2021-07-03T04:11:58.401854Z","iopub.status.idle":"2021-07-03T04:11:58.412601Z","shell.execute_reply.started":"2021-07-03T04:11:58.401782Z","shell.execute_reply":"2021-07-03T04:11:58.411742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.421846Z","iopub.execute_input":"2021-07-03T04:11:58.422277Z","iopub.status.idle":"2021-07-03T04:11:58.436888Z","shell.execute_reply.started":"2021-07-03T04:11:58.422241Z","shell.execute_reply":"2021-07-03T04:11:58.436023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.438271Z","iopub.execute_input":"2021-07-03T04:11:58.438644Z","iopub.status.idle":"2021-07-03T04:11:58.453145Z","shell.execute_reply.started":"2021-07-03T04:11:58.438605Z","shell.execute_reply":"2021-07-03T04:11:58.452283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference(model, states, test_loader, device):\n    model.to(device)\n    tk0 = tqdm(enumerate(test_loader), total=len(test_loader))\n    probs = []  \n    for i, images in tk0:\n        images = images.numpy()\n        images = images.astype(np.float32) / 255\n        images = torch.from_numpy(images)\n        images = images.to(device)\n        #print(images.shape)\n        avg_preds = []\n        for state in states:\n            model.load_state_dict(state['state_dict'])\n            model.eval()\n            with torch.no_grad():\n                y_preds = model(images)\n                probability = F.softmax(y_preds,-1)\n                #print(probability)\n                #print(y_preds)\n            #avg_preds.append(y_preds.sigmoid().to('cpu').numpy())\n            avg_preds.append(probability.to('cpu').numpy())\n        avg_preds = np.mean(avg_preds, axis=0)\n        probs.append(avg_preds)\n    probs = np.concatenate(probs)\n    return probs","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.4544Z","iopub.execute_input":"2021-07-03T04:11:58.454798Z","iopub.status.idle":"2021-07-03T04:11:58.46323Z","shell.execute_reply.started":"2021-07-03T04:11:58.454758Z","shell.execute_reply":"2021-07-03T04:11:58.462373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_multigpu(model_path):\n    state_dict = torch.load(model_path)['model']\n    from collections import OrderedDict\n    new_state_dict = OrderedDict()\n    for k, v in state_dict.items():\n        name = k[7:] # remove `module.`\n        new_state_dict[name] = v\n    return new_state_dict","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.464786Z","iopub.execute_input":"2021-07-03T04:11:58.465353Z","iopub.status.idle":"2021-07-03T04:11:58.473717Z","shell.execute_reply.started":"2021-07-03T04:11:58.465317Z","shell.execute_reply":"2021-07-03T04:11:58.47279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def null_collate(batch):\n    collate = defaultdict(list)\n\n    for r in batch:\n        for k, v in r.items():\n            collate[k].append(v)\n\n    # ---\n    batch_size = len(batch)\n    image = np.stack(collate['image'])\n    image = image.reshape(batch_size, 3, CFG.size,CFG.size)#.repeat(3,1)\n    image = np.ascontiguousarray(image)\n    image = image.astype(np.float32) / 255\n    collate['image'] = torch.from_numpy(image)\n\n    return collate","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.475086Z","iopub.execute_input":"2021-07-03T04:11:58.475486Z","iopub.status.idle":"2021-07-03T04:11:58.484099Z","shell.execute_reply.started":"2021-07-03T04:11:58.475447Z","shell.execute_reply":"2021-07-03T04:11:58.483206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model = CustomModel(CFG.model_name)\nstates = [torch.load(f'../input/covid-models/efficientnetv2_rw_s_512/efficientnetv2_rw_s_512/fold{fold}_model.pth') for fold in CFG.trn_fold]\ntest_dataset = TestDataset(test,study=True, transform=get_transforms(data='valid'))\ntest_loader = DataLoader(test_dataset, batch_size=CFG.batch_size, shuffle=False, \n                         num_workers=CFG.num_workers, pin_memory=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:11:58.485454Z","iopub.execute_input":"2021-07-03T04:11:58.485806Z","iopub.status.idle":"2021-07-03T04:12:10.113079Z","shell.execute_reply.started":"2021-07-03T04:11:58.485768Z","shell.execute_reply":"2021-07-03T04:12:10.112151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = Net(CFG.model_name)\nmodel = Net2()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:10.114387Z","iopub.execute_input":"2021-07-03T04:12:10.114755Z","iopub.status.idle":"2021-07-03T04:12:10.605907Z","shell.execute_reply.started":"2021-07-03T04:12:10.114715Z","shell.execute_reply":"2021-07-03T04:12:10.604986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = inference(model, states, test_loader, device)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:10.607696Z","iopub.execute_input":"2021-07-03T04:12:10.608353Z","iopub.status.idle":"2021-07-03T04:12:12.82132Z","shell.execute_reply.started":"2021-07-03T04:12:10.608311Z","shell.execute_reply":"2021-07-03T04:12:12.82044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:12.822786Z","iopub.execute_input":"2021-07-03T04:12:12.823188Z","iopub.status.idle":"2021-07-03T04:12:12.834632Z","shell.execute_reply.started":"2021-07-03T04:12:12.823142Z","shell.execute_reply":"2021-07-03T04:12:12.833638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.iloc[:,2:] = predictions\n#test[['StudyInstanceUID'] + CFG.target_cols].to_csv(OUTPUT_DIR+'submission.csv', index=False)\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:12.836448Z","iopub.execute_input":"2021-07-03T04:12:12.836812Z","iopub.status.idle":"2021-07-03T04:12:12.856318Z","shell.execute_reply.started":"2021-07-03T04:12:12.836775Z","shell.execute_reply":"2021-07-03T04:12:12.85545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dataset,test_loader,model,predictions,states\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:12.857714Z","iopub.execute_input":"2021-07-03T04:12:12.858163Z","iopub.status.idle":"2021-07-03T04:12:12.875834Z","shell.execute_reply.started":"2021-07-03T04:12:12.858082Z","shell.execute_reply":"2021-07-03T04:12:12.875019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:12.879278Z","iopub.execute_input":"2021-07-03T04:12:12.879596Z","iopub.status.idle":"2021-07-03T04:12:13.038041Z","shell.execute_reply.started":"2021-07-03T04:12:12.879563Z","shell.execute_reply":"2021-07-03T04:12:13.037171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_len","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:13.039443Z","iopub.execute_input":"2021-07-03T04:12:13.039913Z","iopub.status.idle":"2021-07-03T04:12:13.050691Z","shell.execute_reply.started":"2021-07-03T04:12:13.03987Z","shell.execute_reply":"2021-07-03T04:12:13.049694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.columns = ['id', 'PredictionString1', 'negative', 'typical', 'indeterminate', 'atypical']\ndf = pd.merge(df, sub_df, on = 'id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:13.052295Z","iopub.execute_input":"2021-07-03T04:12:13.052714Z","iopub.status.idle":"2021-07-03T04:12:13.067351Z","shell.execute_reply.started":"2021-07-03T04:12:13.052671Z","shell.execute_reply":"2021-07-03T04:12:13.066376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"for i in range(study_len):\n    negative = df.loc[i,'negative']\n    typical = df.loc[i,'typical']\n    indeterminate = df.loc[i,'indeterminate']\n    atypical = df.loc[i,'atypical']\n    df.loc[i, 'PredictionString'] = f'negative {negative} 0 0 1 1 typical {typical} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1'","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:13.068635Z","iopub.execute_input":"2021-07-03T04:12:13.069009Z","iopub.status.idle":"2021-07-03T04:12:13.080609Z","shell.execute_reply.started":"2021-07-03T04:12:13.068971Z","shell.execute_reply":"2021-07-03T04:12:13.079841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study = df[['id', 'PredictionString']]\n\n# df.to_csv('submission.csv',index=False)\n# df","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:13.082034Z","iopub.execute_input":"2021-07-03T04:12:13.082401Z","iopub.status.idle":"2021-07-03T04:12:13.091859Z","shell.execute_reply.started":"2021-07-03T04:12:13.082361Z","shell.execute_reply":"2021-07-03T04:12:13.090888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:13.093163Z","iopub.execute_input":"2021-07-03T04:12:13.09354Z","iopub.status.idle":"2021-07-03T04:12:13.111502Z","shell.execute_reply.started":"2021-07-03T04:12:13.093498Z","shell.execute_reply":"2021-07-03T04:12:13.110426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:13.113158Z","iopub.execute_input":"2021-07-03T04:12:13.113557Z","iopub.status.idle":"2021-07-03T04:12:13.250011Z","shell.execute_reply.started":"2021-07-03T04:12:13.113516Z","shell.execute_reply":"2021-07-03T04:12:13.249036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 class","metadata":{}},{"cell_type":"code","source":"def get_transforms(*, data):\n    \n    if data == 'train':\n        return Compose([\n            Resize(CFG.size, CFG.size),\n           Normalize(\n               mean=[0.485, 0.456, 0.406],\n               std=[0.229, 0.224, 0.225],\n           ),\n            ToTensorV2(),\n        ])\n\n    elif data == 'valid':\n        return Compose([\n            Resize(CFG.size, CFG.size),\n           Normalize(\n               mean=[0.485, 0.456, 0.406],\n               std=[0.229, 0.224, 0.225],\n           ),\n            ToTensorV2(),\n        ])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference(model, states, test_loader, device):\n    model.to(device)\n    tk0 = tqdm(enumerate(test_loader), total=len(test_loader))\n    probs = []  \n    for i, images in tk0:\n\n        images = images.to(device)\n        #print(images.shape)\n        avg_preds = []\n        for state in states:\n            model.load_state_dict(state)\n            model.eval()\n            with torch.no_grad():\n                y_preds = model(images).sigmoid()\n\n            avg_preds.append(y_preds.to('cpu').numpy())\n        avg_preds = np.mean(avg_preds, axis=0)\n        probs.append(avg_preds)\n        \n    probs = np.concatenate(probs)\n    return probs","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:50.537857Z","iopub.execute_input":"2021-07-03T04:14:50.538195Z","iopub.status.idle":"2021-07-03T04:14:50.545432Z","shell.execute_reply.started":"2021-07-03T04:14:50.538165Z","shell.execute_reply":"2021-07-03T04:14:50.544364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomModel2(nn.Module):\n    def __init__(self, model_name='resnet200d_320'):\n        super().__init__()\n        self.model = timm.create_model(model_name, pretrained=False)\n\n        n_features = self.model.classifier.in_features\n        self.model.global_pool = nn.Identity()\n        self.model.classifier = nn.Identity()\n        self.pooling = nn.AdaptiveAvgPool2d(1)\n        self.fc = nn.Linear(n_features, 1)\n\n    def forward(self, x):\n        bs = x.size(0)\n        features = self.model(x)\n        pooled_features = self.pooling(features).view(bs, -1)\n        output = self.fc(pooled_features)\n        return output","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:50.877714Z","iopub.execute_input":"2021-07-03T04:14:50.878089Z","iopub.status.idle":"2021-07-03T04:14:50.88443Z","shell.execute_reply.started":"2021-07-03T04:14:50.878055Z","shell.execute_reply":"2021-07-03T04:14:50.883468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n# sub_df = sub_df.head(200)\nsub_df = sub_df[study_len:]\ntest_paths = f'/kaggle/tmp/{split}/image/' + sub_df['id'] +'.png'\nsub_df['none'] = 0\n\nlabel_cols = sub_df.columns[2]\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:51.312926Z","iopub.execute_input":"2021-07-03T04:14:51.31328Z","iopub.status.idle":"2021-07-03T04:14:51.321075Z","shell.execute_reply.started":"2021-07-03T04:14:51.31325Z","shell.execute_reply":"2021-07-03T04:14:51.319994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = sub_df","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:52.07131Z","iopub.execute_input":"2021-07-03T04:14:52.07167Z","iopub.status.idle":"2021-07-03T04:14:52.07631Z","shell.execute_reply.started":"2021-07-03T04:14:52.071636Z","shell.execute_reply":"2021-07-03T04:14:52.075186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:52.688047Z","iopub.execute_input":"2021-07-03T04:14:52.68839Z","iopub.status.idle":"2021-07-03T04:14:52.698477Z","shell.execute_reply.started":"2021-07-03T04:14:52.688356Z","shell.execute_reply":"2021-07-03T04:14:52.697348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CustomModel2(CFG.model_name2)\nstates = [torch.load(MODEL_DIR2+f'{CFG.model_name2}_fold{fold}_best_loss.pth')['model'] for fold in CFG.trn_fold]\ntest_dataset = TestDataset(test,study=False, transform=get_transforms(data='valid'))\ntest_loader = DataLoader(test_dataset, batch_size=CFG.batch_size, shuffle=False, \n                         num_workers=CFG.num_workers, pin_memory=True)\npredictions = inference(model, states, test_loader, device)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:52.89077Z","iopub.execute_input":"2021-07-03T04:14:52.891156Z","iopub.status.idle":"2021-07-03T04:14:53.20444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:53.983408Z","iopub.execute_input":"2021-07-03T04:14:53.983745Z","iopub.status.idle":"2021-07-03T04:14:54.035279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[label_cols] = predictions","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:14:56.101569Z","iopub.execute_input":"2021-07-03T04:14:56.102018Z","iopub.status.idle":"2021-07-03T04:14:56.169198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:17.852658Z","iopub.execute_input":"2021-07-03T04:12:17.853012Z","iopub.status.idle":"2021-07-03T04:12:17.869694Z","shell.execute_reply.started":"2021-07-03T04:12:17.852983Z","shell.execute_reply":"2021-07-03T04:12:17.868632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_2class = sub_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:17.871315Z","iopub.execute_input":"2021-07-03T04:12:17.871758Z","iopub.status.idle":"2021-07-03T04:12:17.87942Z","shell.execute_reply.started":"2021-07-03T04:12:17.871702Z","shell.execute_reply":"2021-07-03T04:12:17.878574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dataset,test_loader,model,predictions,states\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:17.880662Z","iopub.execute_input":"2021-07-03T04:12:17.881016Z","iopub.status.idle":"2021-07-03T04:12:17.897109Z","shell.execute_reply.started":"2021-07-03T04:12:17.880975Z","shell.execute_reply":"2021-07-03T04:12:17.896107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:17.901859Z","iopub.execute_input":"2021-07-03T04:12:17.902126Z","iopub.status.idle":"2021-07-03T04:12:18.063043Z","shell.execute_reply.started":"2021-07-03T04:12:17.902099Z","shell.execute_reply":"2021-07-03T04:12:18.062146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numba import cuda\nimport torch\ncuda.select_device(0)\ncuda.close()\ncuda.select_device(0)","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:18.066241Z","iopub.execute_input":"2021-07-03T04:12:18.066511Z","iopub.status.idle":"2021-07-03T04:12:19.384839Z","shell.execute_reply.started":"2021-07-03T04:12:18.066483Z","shell.execute_reply":"2021-07-03T04:12:19.384014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# yolov5 predict","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nimport torch","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:19.386118Z","iopub.execute_input":"2021-07-03T04:12:19.386501Z","iopub.status.idle":"2021-07-03T04:12:19.438027Z","shell.execute_reply.started":"2021-07-03T04:12:19.386463Z","shell.execute_reply":"2021-07-03T04:12:19.437169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = meta[meta['split'] == 'test']\nif fast_sub:\n    test_df = fast_df.copy()\nelse:\n    test_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n# test_df = test_df.head(200)\n\n\ntest_df = df[study_len:].reset_index(drop=True) \nmeta['image_id'] = meta['image_id'] + '_image'\nmeta.columns = ['id', 'dim0', 'dim1', 'split']\ntest_df = pd.merge(test_df, meta, on = 'id', how = 'left')\n","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:19.439379Z","iopub.execute_input":"2021-07-03T04:12:19.439728Z","iopub.status.idle":"2021-07-03T04:12:19.451203Z","shell.execute_reply.started":"2021-07-03T04:12:19.439677Z","shell.execute_reply":"2021-07-03T04:12:19.450351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim = 512 #1024, 256, 'original'\ntest_dir = f'/kaggle/tmp/{split}/image'\nweights_dir = '/kaggle/input/siim-cov19-yolov5-train/yolov5/runs/train/exp/weights/best.pt'\n\nshutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\nos.chdir('/kaggle/working/yolov5') # install dependencies\n\nimport torch\n#from IPython.display import Image, clear_output  # to display images\n\n#clear_output()\n#print('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))\n\n\n!python detect.py --weights $weights_dir\\\n--img 512\\\n--conf 0.001\\\n--iou 0.5\\\n--source $test_dir\\\n--save-txt --save-conf --exist-ok\ndef yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n\n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n\n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n\n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n\n    return bboxes\nimage_ids = []\nPredictionStrings = []\n\nfor file_path in tqdm(glob('runs/detect/exp/labels/*.txt')):\n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.id==image_id,['dim1', 'dim0']].values[0]\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 12).astype(str))\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))\n\n\npred_df = pd.DataFrame({'id':image_ids,\n                        'PredictionString':PredictionStrings})","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:19.452651Z","iopub.execute_input":"2021-07-03T04:12:19.453276Z","iopub.status.idle":"2021-07-03T04:12:30.030193Z","shell.execute_reply.started":"2021-07-03T04:12:19.453236Z","shell.execute_reply":"2021-07-03T04:12:30.029139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(['PredictionString'], axis=1)\nsub_df = pd.merge(test_df, pred_df, on = 'id', how = 'left').fillna(\"none 1 0 0 1 1\")\nsub_df = sub_df[['id', 'PredictionString']]\nfor i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] == \"none 1 0 0 1 1\":\n        continue\n    sub_df_split = sub_df.loc[i,'PredictionString'].split()\n    sub_df_list = []\n    for j in range(int(len(sub_df_split) / 6)):\n        sub_df_list.append('opacity')\n        sub_df_list.append(sub_df_split[6 * j + 1])\n        sub_df_list.append(sub_df_split[6 * j + 2])\n        sub_df_list.append(sub_df_split[6 * j + 3])\n        sub_df_list.append(sub_df_split[6 * j + 4])\n        sub_df_list.append(sub_df_split[6 * j + 5])\n    sub_df.loc[i,'PredictionString'] = ' '.join(sub_df_list)\nsub_df['none'] = df_2class['none'] \nfor i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] != 'none 1 0 0 1 1':\n        sub_df.loc[i,'PredictionString'] = sub_df.loc[i,'PredictionString'] + ' none ' + str(sub_df.loc[i,'none']) + ' 0 0 1 1'\nsub_df = sub_df[['id', 'PredictionString']]   \ndf_study = df_study[:study_len]\ndf_study = df_study.append(sub_df).reset_index(drop=True)\ndf_study.to_csv('/kaggle/working/submission.csv',index = False)  \nshutil.rmtree('/kaggle/working/yolov5')","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:30.031831Z","iopub.execute_input":"2021-07-03T04:12:30.032391Z","iopub.status.idle":"2021-07-03T04:12:30.196549Z","shell.execute_reply.started":"2021-07-03T04:12:30.032341Z","shell.execute_reply":"2021-07-03T04:12:30.195667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study","metadata":{"execution":{"iopub.status.busy":"2021-07-03T04:12:30.198741Z","iopub.execute_input":"2021-07-03T04:12:30.199288Z","iopub.status.idle":"2021-07-03T04:12:30.210508Z","shell.execute_reply.started":"2021-07-03T04:12:30.199247Z","shell.execute_reply":"2021-07-03T04:12:30.209606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}