{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-06-26T07:54:30.123994Z","iopub.execute_input":"2021-06-26T07:54:30.124371Z","iopub.status.idle":"2021-06-26T07:55:51.477464Z","shell.execute_reply.started":"2021-06-26T07:54:30.124343Z","shell.execute_reply":"2021-06-26T07:55:51.47616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\nimport torch\nfrom torch import nn\nimport cv2\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-26T07:55:51.479897Z","iopub.execute_input":"2021-06-26T07:55:51.480296Z","iopub.status.idle":"2021-06-26T07:55:54.778762Z","shell.execute_reply.started":"2021-06-26T07:55:51.480251Z","shell.execute_reply":"2021-06-26T07:55:54.777647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nfast_sub = False\n\nif df.shape[0] == 2477:\n    fast_sub = True\n    fast_df = pd.DataFrame(([['00086460a852_study', 'negative 1 0 0 1 1'], \n                         ['000c9c05fd14_study', 'negative 1 0 0 1 1'], \n                         ['65761e66de9f_image', 'none 1 0 0 1 1'], \n                         ['51759b5579bc_image', 'none 1 0 0 1 1']]), \n                       columns=['id', 'PredictionString'])\nelse:\n    fast_sub = False\n","metadata":{"execution":{"iopub.status.busy":"2021-06-26T07:55:54.781082Z","iopub.execute_input":"2021-06-26T07:55:54.781432Z","iopub.status.idle":"2021-06-26T07:55:54.802524Z","shell.execute_reply.started":"2021-06-26T07:55:54.781402Z","shell.execute_reply":"2021-06-26T07:55:54.801516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# .dcm to .png","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-06-26T07:55:54.80622Z","iopub.execute_input":"2021-06-26T07:55:54.806548Z","iopub.status.idle":"2021-06-26T07:55:55.082028Z","shell.execute_reply.started":"2021-06-26T07:55:54.80652Z","shell.execute_reply":"2021-06-26T07:55:55.080958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-06-26T07:55:55.083658Z","iopub.execute_input":"2021-06-26T07:55:55.084064Z","iopub.status.idle":"2021-06-26T07:55:55.093522Z","shell.execute_reply.started":"2021-06-26T07:55:55.084024Z","shell.execute_reply":"2021-06-26T07:55:55.09231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsplit = 'test'\nsave_dir = f'/kaggle/tmp/{split}/'\n\nos.makedirs(save_dir, exist_ok=True)\n\nsave_dir = f'/kaggle/tmp/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=600)  \n    study = '00086460a852' + '_study.png'\n    im.save(os.path.join(save_dir, study))\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=600)  \n    study = '000c9c05fd14' + '_study.png'\n    im.save(os.path.join(save_dir, study))\nelse:   \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=600)  \n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))\n","metadata":{"execution":{"iopub.status.busy":"2021-06-26T07:55:55.097259Z","iopub.execute_input":"2021-06-26T07:55:55.097669Z","iopub.status.idle":"2021-06-26T08:06:24.660745Z","shell.execute_reply.started":"2021-06-26T07:55:55.097606Z","shell.execute_reply":"2021-06-26T08:06:24.659592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\nsave_dir = f'/kaggle/tmp/{split}/image/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=512)  \n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)\nmeta = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:06:24.662478Z","iopub.execute_input":"2021-06-26T08:06:24.663082Z","iopub.status.idle":"2021-06-26T08:16:28.131603Z","shell.execute_reply.started":"2021-06-26T08:06:24.663038Z","shell.execute_reply":"2021-06-26T08:16:28.130396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study predict","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nif fast_sub:\n    df = fast_df.copy()\nelse:\n    df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\n\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:16:28.13453Z","iopub.execute_input":"2021-06-26T08:16:28.135247Z","iopub.status.idle":"2021-06-26T08:16:28.199987Z","shell.execute_reply.started":"2021-06-26T08:16:28.135185Z","shell.execute_reply":"2021-06-26T08:16:28.198876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['id_last_str']","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:03.523513Z","iopub.execute_input":"2021-06-26T08:18:03.523979Z","iopub.status.idle":"2021-06-26T08:18:03.536485Z","shell.execute_reply.started":"2021-06-26T08:18:03.523947Z","shell.execute_reply":"2021-06-26T08:18:03.535078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_len","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:04.362178Z","iopub.execute_input":"2021-06-26T08:18:04.362785Z","iopub.status.idle":"2021-06-26T08:18:04.371476Z","shell.execute_reply.started":"2021-06-26T08:18:04.362753Z","shell.execute_reply":"2021-06-26T08:18:04.370177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512)\nDIM = (512,512,3)\n\n#load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\nif fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[:study_len]\ntest_paths = f'/kaggle/tmp/{split}/study/' + sub_df['id'] +'.png'\n\nsub_df['negative'] = 0\nsub_df['typical'] = 0\nsub_df['indeterminate'] = 0\nsub_df['atypical'] = 0\n\n\nlabel_cols = sub_df.columns[2:]\n\nsub_df['test_path'] = test_paths","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:09.745734Z","iopub.execute_input":"2021-06-26T08:18:09.746071Z","iopub.status.idle":"2021-06-26T08:18:09.869267Z","shell.execute_reply.started":"2021-06-26T08:18:09.746042Z","shell.execute_reply":"2021-06-26T08:18:09.868268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('../input/timm-pytorch-image-models/pytorch-image-models-master')\nimport timm\n\nclass SiimClfModel(nn.Module):\n    def __init__(self, backbone_name, backbone_pretrained, n_classes=4):\n        super(SiimClfModel, self).__init__()\n        self.backbone = timm.create_model(backbone_name, pretrained=False)\n        if(backbone_pretrained):\n            self.backbone.load_state_dict(torch.load(backbone_pretrained))\n        clf_in_feature = self.backbone.classifier.in_features\n        self.backbone.classifier = nn.Linear(clf_in_feature, n_classes)\n        \n    def forward(self, x):\n        batch_size = x.shape[0]\n        x = self.backbone(x)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:13.227893Z","iopub.execute_input":"2021-06-26T08:18:13.228313Z","iopub.status.idle":"2021-06-26T08:18:15.366235Z","shell.execute_reply.started":"2021-06-26T08:18:13.228283Z","shell.execute_reply":"2021-06-26T08:18:15.364976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ClfTestDataset(torch.utils.data.Dataset):\n    \n    def __init__(self, csv, transforms=None):\n        self.csv = csv.reset_index()\n        self.augmentations = transforms\n\n    def __len__(self):\n        return self.csv.shape[0]\n\n    def __getitem__(self, index):\n        row = self.csv.iloc[index]\n        image = cv2.imread(row.test_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        if self.augmentations:\n            augmented = self.augmentations(image=image)\n            image = augmented['image']\n        \n        if('clf_label_idx' in self.csv.columns):\n            return torch.tensor(image), torch.tensor(row.clf_label_idx)\n        \n        return image\n    \n\ndef get_test_transforms(candidate):\n    dim = candidate.get('dim', DIM)\n    return A.Compose(\n        [\n            A.Resize(dim[0],dim[1],always_apply=True),\n            A.Normalize(),\n            ToTensorV2(p=1.0)\n        ]\n    )","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:17.517876Z","iopub.execute_input":"2021-06-26T08:18:17.518208Z","iopub.status.idle":"2021-06-26T08:18:17.530316Z","shell.execute_reply.started":"2021-06-26T08:18:17.518177Z","shell.execute_reply":"2021-06-26T08:18:17.528717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CANDIDATES = [\n    {\n        'backbone_name':'efficientnet_b3',\n        'model_path':'../input/siimstudyclfmodels/v2/Fold0_efficientnet_b3_v2_5folds_ValidLoss0.962_ValidAcc0.624_ValidAUC0.708_Ep01.pth'\n    },\n    {\n        'backbone_name':'efficientnet_b3',\n        'model_path':'../input/siimstudyclfmodels/v2/Fold1_efficientnet_b3_v2_5folds_ValidLoss0.931_ValidAcc0.644_ValidAUC0.726_Ep01.pth'\n    },\n    {\n        'backbone_name':'efficientnet_b3',\n        'model_path':'../input/siimstudyclfmodels/v2/Fold2_efficientnet_b3_v2_5folds_ValidLoss0.921_ValidAcc0.634_ValidAUC0.753_Ep02.pth'\n    },\n    {\n        'backbone_name':'efficientnet_b3',\n        'model_path':'../input/siimstudyclfmodels/v2/Fold3_efficientnet_b3_v2_5folds_ValidLoss0.943_ValidAcc0.626_ValidAUC0.724_Ep01.pth'\n    },\n    {\n        'backbone_name':'efficientnet_b3',\n        'model_path':'../input/siimstudyclfmodels/v2/Fold4_efficientnet_b3_v2_5folds_ValidLoss0.955_ValidAcc0.635_ValidAUC0.706_Ep01.pth'\n    },\n]\n","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:23.040911Z","iopub.execute_input":"2021-06-26T08:18:23.041294Z","iopub.status.idle":"2021-06-26T08:18:23.047554Z","shell.execute_reply.started":"2021-06-26T08:18:23.041265Z","shell.execute_reply":"2021-06-26T08:18:23.045807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:18:23.315475Z","iopub.execute_input":"2021-06-26T08:18:23.315854Z","iopub.status.idle":"2021-06-26T08:18:23.322696Z","shell.execute_reply.started":"2021-06-26T08:18:23.315826Z","shell.execute_reply":"2021-06-26T08:18:23.319408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm \n\ndef clf_predict_fn(dataloader, model, device):\n    model.eval()\n    tk0 = tqdm(enumerate(dataloader), total=len(dataloader))\n    batch_preds=[]\n    \n    for i, inps in tk0:\n        inps = inps.to(device)\n        outputs = model(inps)\n        probs = nn.functional.softmax(outputs, dim=-1)\n        batch_preds.append(probs.detach().cpu().numpy())\n        \n        del inps, outputs, probs\n        torch.cuda.empty_cache()\n        \n    return np.concatenate(batch_preds)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:19:23.211301Z","iopub.execute_input":"2021-06-26T08:19:23.211794Z","iopub.status.idle":"2021-06-26T08:19:23.221725Z","shell.execute_reply.started":"2021-06-26T08:19:23.211761Z","shell.execute_reply":"2021-06-26T08:19:23.220441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for candidate in CANDIDATES:\n    model = SiimClfModel(backbone_name=candidate['backbone_name'], backbone_pretrained=None)\n    model.load_state_dict(torch.load(candidate['model_path'], map_location='cpu'))\n    model.to('cuda:0')\n    print()\n    \n    test_dataset = ClfTestDataset(sub_df, get_test_transforms(candidate))\n    test_dataloader = torch.utils.data.DataLoader(test_dataset, shuffle=False, batch_size=BATCH_SIZE)\n    \n    sub_df[label_cols] += clf_predict_fn(test_dataloader, model, torch.device('cuda:0'))\n    \n    del model\n    torch.cuda.empty_cache()\n    \nsub_df[label_cols] /= len(CANDIDATES)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T08:19:23.606133Z","iopub.execute_input":"2021-06-26T08:19:23.606486Z","iopub.status.idle":"2021-06-26T08:19:24.151057Z","shell.execute_reply.started":"2021-06-26T08:19:23.606446Z","shell.execute_reply":"2021-06-26T08:19:24.147822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.columns = ['id', 'PredictionString1', 'negative', 'typical', 'indeterminate', 'atypical', 'test_path']\ndf = pd.merge(df, sub_df, on = 'id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2021-06-26T05:52:56.618074Z","iopub.execute_input":"2021-06-26T05:52:56.618628Z","iopub.status.idle":"2021-06-26T05:52:56.627889Z","shell.execute_reply.started":"2021-06-26T05:52:56.61859Z","shell.execute_reply":"2021-06-26T05:52:56.62707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"for i in range(study_len):\n    negative = df.loc[i,'negative']\n    typical = df.loc[i,'typical']\n    indeterminate = df.loc[i,'indeterminate']\n    atypical = df.loc[i,'atypical']\n    df.loc[i, 'PredictionString'] = f'negative {negative} 0 0 1 1 typical {typical} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1'","metadata":{"execution":{"iopub.status.busy":"2021-06-26T05:52:56.629433Z","iopub.execute_input":"2021-06-26T05:52:56.629989Z","iopub.status.idle":"2021-06-26T05:52:56.642111Z","shell.execute_reply.started":"2021-06-26T05:52:56.62995Z","shell.execute_reply":"2021-06-26T05:52:56.641286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study = df[['id', 'PredictionString']]\n\n# df.to_csv('submission.csv',index=False)\n# df","metadata":{"execution":{"iopub.status.busy":"2021-06-26T05:52:56.645018Z","iopub.execute_input":"2021-06-26T05:52:56.645607Z","iopub.status.idle":"2021-06-26T05:52:56.655342Z","shell.execute_reply.started":"2021-06-26T05:52:56.645567Z","shell.execute_reply":"2021-06-26T05:52:56.653658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = df_study.fillna('')\ndef remove_prediction_on_image_level(row):\n    if(row['id'].endswith('_image')):\n        row['PredictionString'] = ''\n    return row\nsub_df = sub_df.apply(remove_prediction_on_image_level, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T05:52:56.657866Z","iopub.execute_input":"2021-06-26T05:52:56.658215Z","iopub.status.idle":"2021-06-26T05:52:56.673797Z","shell.execute_reply.started":"2021-06-26T05:52:56.658182Z","shell.execute_reply":"2021-06-26T05:52:56.672753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.iloc[0].PredictionString","metadata":{"execution":{"iopub.status.busy":"2021-06-26T05:52:56.677539Z","iopub.execute_input":"2021-06-26T05:52:56.679216Z","iopub.status.idle":"2021-06-26T05:52:56.688472Z","shell.execute_reply.started":"2021-06-26T05:52:56.679179Z","shell.execute_reply":"2021-06-26T05:52:56.68752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T05:52:56.695013Z","iopub.execute_input":"2021-06-26T05:52:56.696945Z","iopub.status.idle":"2021-06-26T05:52:56.706487Z","shell.execute_reply.started":"2021-06-26T05:52:56.696905Z","shell.execute_reply":"2021-06-26T05:52:56.705433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}