{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Pre-processing","metadata":{}},{"cell_type":"code","source":"# Install pydicom\n!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:16:57.152155Z","iopub.execute_input":"2021-07-26T13:16:57.152567Z","iopub.status.idle":"2021-07-26T13:18:05.060276Z","shell.execute_reply.started":"2021-07-26T13:16:57.152519Z","shell.execute_reply":"2021-07-26T13:18:05.059274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import packages\nimport os\nimport glob\nimport cv2\nimport pickle\nimport json\nimport numpy as np\nimport pandas as pd\n\nfrom pathlib import Path\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom tqdm.auto import tqdm as tqdm_auto\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:05.063894Z","iopub.execute_input":"2021-07-26T13:18:05.064203Z","iopub.status.idle":"2021-07-26T13:18:05.492173Z","shell.execute_reply.started":"2021-07-26T13:18:05.064173Z","shell.execute_reply":"2021-07-26T13:18:05.491338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Args\ndata_dir = Path('/kaggle/input/siim-covid19-detection')\nsave_dir = Path('/kaggle/working/data/siim')\ntest_csv_path = data_dir / 'sample_submission.csv'\ntest_dcm_dir = data_dir / 'test'\nimage_size = 2048\ndebug = False","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:05.493771Z","iopub.execute_input":"2021-07-26T13:18:05.494089Z","iopub.status.idle":"2021-07-26T13:18:05.4981Z","shell.execute_reply.started":"2021-07-26T13:18:05.494059Z","shell.execute_reply":"2021-07-26T13:18:05.497354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read csv\nprint(f'Read csv from {test_csv_path}')\nsub = pd.read_csv(test_csv_path)\nstudy_sub = sub[sub.id.apply(lambda x: x.endswith('study'))]  # Study-only dataframe\nimage_sub = sub[sub.id.apply(lambda x: x.endswith('image'))]  # Study-only dataframe\n\n# Read dicom file list\nprint(f'Read dcm from {test_dcm_dir}')\ndcm_paths = list(test_dcm_dir.glob('**/*.dcm'))","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:05.499602Z","iopub.execute_input":"2021-07-26T13:18:05.499938Z","iopub.status.idle":"2021-07-26T13:18:09.825767Z","shell.execute_reply.started":"2021-07-26T13:18:05.499902Z","shell.execute_reply":"2021-07-26T13:18:09.824821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Subsample for debugging\nif debug:\n    dcm_paths = dcm_paths[:10]\n    studies = list(set([str(d).split('/')[-3] + '_study' for d in dcm_paths]))\n    images = list(set([str(d).split('/')[-2] + '_image' for d in dcm_paths]))\n    study_sub = study_sub[study_sub.id.isin(studies)]\n    image_sub = image_sub[image_sub.id.isin(images)]","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:09.82703Z","iopub.execute_input":"2021-07-26T13:18:09.827425Z","iopub.status.idle":"2021-07-26T13:18:09.838378Z","shell.execute_reply.started":"2021-07-26T13:18:09.827389Z","shell.execute_reply":"2021-07-26T13:18:09.835977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make directories to save jpg\njpg_dir = save_dir / f'{image_size}x{image_size}'\nprint(f'Create directory to save jpg: {jpg_dir}')\njpg_dir.mkdir(parents=True, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:09.839713Z","iopub.execute_input":"2021-07-26T13:18:09.840064Z","iopub.status.idle":"2021-07-26T13:18:09.937533Z","shell.execute_reply.started":"2021-07-26T13:18:09.840028Z","shell.execute_reply":"2021-07-26T13:18:09.93645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut=True, fix_monochrome=True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\n\ndef resize_xray(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:09.939135Z","iopub.execute_input":"2021-07-26T13:18:09.939529Z","iopub.status.idle":"2021-07-26T13:18:09.947923Z","shell.execute_reply.started":"2021-07-26T13:18:09.939493Z","shell.execute_reply":"2021-07-26T13:18:09.946774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert dcm to jpg\nimage_ids = []\nfolder_ids = []\nstudy_ids = []\nwidths = []\nheights = []\nfor path in tqdm(dcm_paths, total=len(dcm_paths)):\n    path_split = str(path).split('/')\n    study_id = path_split[-3]\n    folder_id = path_split[-2]\n    image_name = path_split[-1].replace('.dcm', '')\n            \n    xray = read_xray(path)\n    im_path = save_dir / f'{image_size}x{image_size}' / (image_name + '.jpg')\n    if not im_path.exists():\n        im = resize_xray(xray, size=image_size)\n        im.save(im_path)\n    \n    image_ids.append(image_name)\n    folder_ids.append(folder_id)\n    study_ids.append(study_id)\n    widths.append(xray.shape[1])\n    heights.append(xray.shape[0])\n\ntest_df = pd.DataFrame.from_dict({'id': image_ids,\n                                  'folder_id':folder_ids,\n                                  'StudyInstanceUID': study_ids,\n                                  'width': widths,\n                                  'height': heights})\ntest_df.to_csv(save_dir / 'test_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:09.950557Z","iopub.execute_input":"2021-07-26T13:18:09.950944Z","iopub.status.idle":"2021-07-26T13:18:19.32814Z","shell.execute_reply.started":"2021-07-26T13:18:09.950908Z","shell.execute_reply":"2021-07-26T13:18:19.327008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert meta to mmdet format\nanno_dir = save_dir / 'mmdet_annos'\nanno_path = anno_dir / 'test.json'\n\nmmdet_anno = {\n  \"info\": {\n    \"description\": \"SIIM Covid-19\",\n    \"year\": 2021,\n    \"url\": None,\n    \"version\": None,\n    \"contributor\": None,\n  },\n  \"licenses\": [\n    {\n      \"id\": 0,\n      \"url\": None,\n      \"name\": None\n    }\n  ],\n  \"categories\": [\n    {\n      \"id\": 0,\n      \"supercategory\": None,\n      \"name\": \"opacity\"\n    }\n  ],\n  \"type\": \"instances\",\n  \"annotations\": [],\n  \"images\": []\n}\n\nfor idx, row in test_df.iterrows():\n    image = {\n        'license': 0,\n        'url': None,\n        'file_name': row.id + '.jpg',\n        'height': image_size,\n        'width': image_size,\n        'date_captured': None,\n        'id': idx\n    }\n    mmdet_anno['images'].append(image)\n\nanno_dir.mkdir(parents=True, exist_ok=True)\nprint(f'Save mmdet anno to {anno_path}')\njson.dump(mmdet_anno, open(anno_path, 'w'), indent=2)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:19.329879Z","iopub.execute_input":"2021-07-26T13:18:19.330262Z","iopub.status.idle":"2021-07-26T13:18:19.342169Z","shell.execute_reply.started":"2021-07-26T13:18:19.330223Z","shell.execute_reply":"2021-07-26T13:18:19.341072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"# Install mmdetection packages\n%cd /kaggle/input/mmdetection-swin-env/mmdetection_swin\n!pip install --no-index -f wheels/mmcv-full/ mmcv-full\n!pip install --no-index -f wheels/runtime/ -r requirements/runtime.txt\n!pip install \".[runtime]\"\n!pip install --no-index -f wheels/ensemble-boxes/ ensemble-boxes\n\nimport ensemble_boxes","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:18:19.343642Z","iopub.execute_input":"2021-07-26T13:18:19.344174Z","iopub.status.idle":"2021-07-26T13:19:42.546634Z","shell.execute_reply.started":"2021-07-26T13:18:19.344107Z","shell.execute_reply":"2021-07-26T13:19:42.545494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run test\nconfig = Path('configs/siim/swin_kaggle_many.py')\nout_dir = Path('/kaggle/working/output')\ncheckpoints = [\n    '/kaggle/input/swin-checkpoints/swin_v3_expt8_fold0/epoch_30.pth',\n    '/kaggle/input/swin-checkpoints/swin_v3_expt8_fold1/epoch_20.pth',\n    '/kaggle/input/swin-checkpoints/swin_v3_expt8_fold2/epoch_15.pth',\n    '/kaggle/input/swin-checkpoints/swin_v3_expt8_fold3/epoch_27.pth',\n    '/kaggle/input/swin-checkpoints/swin_v3_expt8_fold4/epoch_22.pth',\n]\nout_pkls = [out_dir / f'result_{n}.pkl' for n in range(len(checkpoints))]\n\nfor ckpt, out_pkl in zip(checkpoints, out_pkls):\n    out_dir.mkdir(parents=True, exist_ok=True)\n    test_cmd = f'python tools/test.py {config} {ckpt} --out {out_pkl}'\n    print(test_cmd)\n    !{test_cmd}","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:19:42.548286Z","iopub.execute_input":"2021-07-26T13:19:42.548659Z","iopub.status.idle":"2021-07-26T13:23:01.327893Z","shell.execute_reply.started":"2021-07-26T13:19:42.548609Z","shell.execute_reply":"2021-07-26T13:23:01.323348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize for debugging\nif debug:\n    import time\n    from IPython.display import clear_output\n    import matplotlib.pyplot as plt\n    %matplotlib inline\n\n    threshold = 0.3\n    visual_size = 512\n    colors = [[255, 0, 0], [0,255,0], [0,0,255], [255,0,255], [255,255,0]]\n    \n    def bbox_to_contour(bbox):\n        contour = [\n            [bbox[0], bbox[1]],\n            [bbox[2], bbox[1]],\n            [bbox[2], bbox[3]],\n            [bbox[0], bbox[3]],\n            [bbox[0], bbox[1]],\n        ]\n        contour = np.array(contour, dtype=int)\n        return contour\n\n    annotations = json.load(open(anno_path, 'r'))\n    images = {v['id']: v['file_name'] for v in annotations['images']}\n    \n    preds = {k: [] for k in images.keys()}\n    scores = {k: [] for k in images.keys()}\n    for out_pkl in out_pkls:\n        predictions = pickle.load(open(out_pkl, 'rb'))\n        for i, pred in enumerate(predictions):\n            preds[i].append(pred[0][0][:, :4])\n            scores[i].append(pred[0][0][:, -1])\n\n    for i in images.keys():\n        image_path = os.path.join(jpg_dir, images[i])\n        image_org = cv2.imread(image_path)\n        image = cv2.resize(image_org, (visual_size, visual_size))\n        \n        # raw results\n        for k, (pred, score) in enumerate(zip(preds[i], scores[i])):\n            pred = (np.array(pred, dtype=float) * visual_size / image_size).astype(int)\n            score = np.array(score)\n            for pr, sc in zip(pred[score > threshold], score[score > threshold]):\n                cv2.drawContours(image, [bbox_to_contour(pr)], -1, colors[k], 3)\n                cv2.putText(image, f'{sc:.4f}', pr[:2] - np.array([0, 10]), cv2.FONT_HERSHEY_SIMPLEX, 1, colors[k], 2, cv2.LINE_AA)\n\n        clear_output(wait=True)\n        plt.imshow(image)\n        plt.show()\n        \n        # ensemble\n        iou_thr = 0.5\n        skip_box_thr = 0.0001\n        \n        boxes_list = [x / image_size for x in preds[i]]\n        scores_list = [x for x in scores[i]]\n        labels_list = [[0] * len(x) for x in scores[i]]\n        weights = [1 for x in scores[i]]\n\n        boxes_ens, scores_ens, _ = ensemble_boxes.weighted_boxes_fusion(boxes_list, \n                                                                        scores_list, \n                                                                        labels_list, \n                                                                        weights=weights, \n                                                                        iou_thr=iou_thr, \n                                                                        skip_box_thr=skip_box_thr,\n                                                                        conf_type='max')\n        image = cv2.resize(image_org, (visual_size, visual_size))\n        pred = (np.array(boxes_ens, dtype=float) * visual_size).astype(int)\n        score = np.array(scores_ens)\n\n        for pr, sc in zip(pred[score > threshold], score[score > threshold]):\n            cv2.drawContours(image, [bbox_to_contour(pr)], -1, colors[0], 3)\n            cv2.putText(image, f'{sc:.4f}', pr[:2] - np.array([0, 10]), cv2.FONT_HERSHEY_SIMPLEX, 1, colors[0], 2, cv2.LINE_AA)\n        plt.imshow(image)\n        plt.show()\n        time.sleep(1)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:23:01.336018Z","iopub.execute_input":"2021-07-26T13:23:01.336355Z","iopub.status.idle":"2021-07-26T13:23:15.496864Z","shell.execute_reply.started":"2021-07-26T13:23:01.336318Z","shell.execute_reply":"2021-07-26T13:23:15.495937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Post-processing","metadata":{}},{"cell_type":"code","source":"# Parse result pickles\nannotations = json.load(open(anno_path, 'r'))\nimages = {v['id']: v['file_name'].split('.')[0] for v in annotations['images']}\n\nbboxes = {k: [] for k in images.values()}\nscores = {k: [] for k in images.values()}\nfor out_pkl in out_pkls:\n    predictions = pickle.load(open(out_pkl, 'rb'))\n    for i, pred in enumerate(predictions):\n        image_name = images[i]\n        bboxes[image_name].append(pred[0][0][:, :4])\n        scores[image_name].append(pred[0][0][:, -1])","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:23:15.498266Z","iopub.execute_input":"2021-07-26T13:23:15.498631Z","iopub.status.idle":"2021-07-26T13:23:15.514751Z","shell.execute_reply.started":"2021-07-26T13:23:15.498591Z","shell.execute_reply":"2021-07-26T13:23:15.51396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensemble\nensemble = 'wbf'\n\nif ensemble == 'wbf':\n    bboxes_final, scores_final = {}, {}\n    for k in images.values():\n        bboxes_ = [x / image_size for x in bboxes[k]]\n        scores_ = scores[k]\n        labels_ = [[0] * len(x) for x in scores[k]]\n        weights_ = [1 for x in scores[k]]\n\n        bboxes_, scores_, _ = ensemble_boxes.weighted_boxes_fusion(bboxes_,\n                                                                   scores_,\n                                                                   labels_,\n                                                                   weights=weights_,\n                                                                   iou_thr=0.5,\n                                                                   skip_box_thr= 0.0001,\n                                                                   )\n        bboxes_final[k] = bboxes_ * image_size\n        scores_final[k] = scores_\n        \nelif ensemble == 'concat':\n    bboxes_final, scores_final = {}, {}\n    for k in images.values():\n        bboxes_final[k] = np.concatenate(bboxes[k])\n        scores_final[k] = np.concatenate(scores[k])\n        \nelse:\n    raise","metadata":{"execution":{"iopub.status.busy":"2021-07-26T13:24:59.589839Z","iopub.execute_input":"2021-07-26T13:24:59.590206Z","iopub.status.idle":"2021-07-26T13:24:59.598359Z","shell.execute_reply.started":"2021-07-26T13:24:59.590168Z","shell.execute_reply":"2021-07-26T13:24:59.597239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make submission file\nsubmission = []\nfor idx, row in test_df.iterrows():\n    image_id = row['id']\n    width = row['width']\n    height = row['height']\n    \n    bbox = bboxes_final[image_id]\n    score = scores_final[image_id]\n    \n    bbox[:, 0] = bbox[:, 0] * width / image_size\n    bbox[:, 1] = bbox[:, 1] * height / image_size\n    bbox[:, 2] = bbox[:, 2] * width / image_size\n    bbox[:, 3] = bbox[:, 3] * height / image_size\n    bbox = bbox.astype(int)\n    \n    pred_str = ' '.join(['opacity {:f} {:d} {:d} {:d} {:d}'.format(s, *b) for s, b in zip(score, bbox)])\n    case = {'Id': image_id + '_image',\n            'PredictionString': pred_str}\n    submission.append(case)\n\nfor idx, row in study_sub.iterrows():\n    case = {'Id': row['id'], \n            'PredictionString': ''}\n    submission.append(case)\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df","metadata":{"_kg_hide-output":true,"papermill":{"duration":2.152994,"end_time":"2021-07-06T18:01:16.354583","exception":false,"start_time":"2021-07-06T18:01:14.201589","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-26T13:26:22.589688Z","iopub.execute_input":"2021-07-26T13:26:22.590034Z","iopub.status.idle":"2021-07-26T13:26:22.623457Z","shell.execute_reply.started":"2021-07-26T13:26:22.590002Z","shell.execute_reply":"2021-07-26T13:26:22.62252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"_kg_hide-output":true,"papermill":{"duration":2.152994,"end_time":"2021-07-06T18:01:16.354583","exception":false,"start_time":"2021-07-06T18:01:14.201589","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-26T13:26:38.089136Z","iopub.execute_input":"2021-07-26T13:26:38.089496Z","iopub.status.idle":"2021-07-26T13:26:38.09972Z","shell.execute_reply.started":"2021-07-26T13:26:38.089465Z","shell.execute_reply":"2021-07-26T13:26:38.098572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf data output","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}