{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r ../input/siim-covid-packages .\n!mv ./siim-covid-packages/efficientnet-pytorch-0.7.0-pyhd8ed1ab_0.tar.xyz ./siim-covid-packages/efficientnet-pytorch-0.7.0-pyhd8ed1ab_0.tar.bz2\n!mv ./siim-covid-packages/pretrainedmodels-0.7.4-py37hc8dfbb8_0.tar.xyz ./siim-covid-packages/pretrainedmodels-0.7.4-py37hc8dfbb8_0.tar.bz2\n!mv ./siim-covid-packages/antlr4-python3-runtime-4.8.tar.xyz ./siim-covid-packages/antlr4-python3-runtime-4.8.tar.gz\n!mv ./siim-covid-packages/pycocotools-2.0.2.tar.xyz ./siim-covid-packages/pycocotools-2.0.2.tar.gz\n\n!pip install ./siim-covid-packages/python_gdcm-3.0.9.0-cp37-cp37m-manylinux2014_x86_64.whl\n!conda install ./siim-covid-packages/efficientnet-pytorch-0.7.0-pyhd8ed1ab_0.tar.bz2\n!conda install ./siim-covid-packages/pretrainedmodels-0.7.4-py37hc8dfbb8_0.tar.bz2\n!pip install ./siim-covid-packages/timm-0.4.5-py3-none-any.whl\n!pip install ./siim-covid-packages/antlr4-python3-runtime-4.8.tar.gz\n!pip install ./siim-covid-packages/pycocotools-2.0.2.tar.gz\n!pip install ./siim-covid-packages/omegaconf-2.0.6-py3-none-any.whl\n!pip install ./siim-covid-packages/ensemble_boxes-1.0.6-py3-none-any.whl\n!rm -rf ./siim-covid-packages","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:52:13.841459Z","iopub.execute_input":"2021-07-31T12:52:13.841996Z","iopub.status.idle":"2021-07-31T12:55:40.474469Z","shell.execute_reply.started":"2021-07-31T12:52:13.841888Z","shell.execute_reply":"2021-07-31T12:55:40.473457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nimport pydicom\nimport cv2\nimport torch\nimport gc\nimport pickle\nfrom tqdm import tqdm\nimport random\nfrom ensemble_boxes import weighted_boxes_fusion\nfrom multiprocessing import Pool\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:55:40.476127Z","iopub.execute_input":"2021-07-31T12:55:40.476474Z","iopub.status.idle":"2021-07-31T12:55:43.085909Z","shell.execute_reply.started":"2021-07-31T12:55:40.476429Z","shell.execute_reply":"2021-07-31T12:55:43.085045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dict(name):\n    with open(name, 'rb') as f:\n        return pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:55:43.087688Z","iopub.execute_input":"2021-07-31T12:55:43.088008Z","iopub.status.idle":"2021-07-31T12:55:43.094991Z","shell.execute_reply.started":"2021-07-31T12:55:43.087981Z","shell.execute_reply":"2021-07-31T12:55:43.094271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = [\n    'Negative for Pneumonia',\n    'Typical Appearance',\n    'Indeterminate Appearance',\n    'Atypical Appearance'\n]\n\nstudy_submission_classes = {\n    'Negative for Pneumonia': 'negative',\n    'Typical Appearance': 'typical',\n    'Indeterminate Appearance': 'indeterminate',\n    'Atypical Appearance': 'atypical'\n}","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:55:43.096471Z","iopub.execute_input":"2021-07-31T12:55:43.096822Z","iopub.status.idle":"2021-07-31T12:55:43.103827Z","shell.execute_reply.started":"2021-07-31T12:55:43.096769Z","shell.execute_reply":"2021-07-31T12:55:43.102959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### just pick 14 study in public test set intead of 1214 study to save time, don't need to run 1214 study\n#### now just run kernel on private test study + 14 public test study\n\npublic_test_meta_df = pd.read_csv('../input/siim-covid-public-test/test_meta.csv')\npublic_test_14_study = list(np.unique(public_test_meta_df.studyid.values))[0:14]\npublic_test_1200_study = []\n\npublic_test_submisison_df = pd.read_csv('../input/siim-covid-public-test/submission_0.658_20210731.csv')\npublic_test_1200_study_level_output = []\npublic_test_1200_image_level_output = []\nfor studyid, grp in tqdm(public_test_meta_df.groupby('studyid')):\n    if studyid in public_test_14_study:\n        continue\n        \n    public_test_1200_study.append(studyid)\n    \n    study_tmp_df = public_test_submisison_df.loc[public_test_submisison_df['id'] == '{}_study'.format(studyid)]\n    assert len(study_tmp_df) == 1\n    public_test_1200_study_level_output.append(['{}_study'.format(studyid), study_tmp_df.PredictionString.values[0]])\n\n    for _, row in grp.iterrows():\n        image_tmp_df = public_test_submisison_df.loc[public_test_submisison_df['id'] == '{}_image'.format(row['imageid'])]\n        assert len(image_tmp_df) == 1\n        public_test_1200_image_level_output.append(['{}_image'.format(row['imageid']), image_tmp_df.PredictionString.values[0]])\n\npublic_test_1200_output = public_test_1200_study_level_output + public_test_1200_image_level_output\npublic_test_1200_submission_df = pd.DataFrame(data=np.array(public_test_1200_output), columns=['id','PredictionString'])\npublic_test_1200_submission_df.to_csv('./submission_1200_study.csv', index=False)\nprint(public_test_1200_submission_df.shape, len(public_test_14_study), len(public_test_1200_study))\n\ndel public_test_1200_output\ndel public_test_1200_study_level_output\ndel public_test_1200_image_level_output\ndel public_test_submisison_df\ndel public_test_1200_submission_df\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:55:43.105002Z","iopub.execute_input":"2021-07-31T12:55:43.105363Z","iopub.status.idle":"2021-07-31T12:55:48.606671Z","shell.execute_reply.started":"2021-07-31T12:55:43.105326Z","shell.execute_reply":"2021-07-31T12:55:48.605862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### extract dicom to image\nos.makedirs('./images', exist_ok=True)\nos.makedirs('./csv', exist_ok=True)\n\nclass ME:\n    def __init__(self, StudyInstanceUID, file_path):\n        self.StudyInstanceUID = StudyInstanceUID\n        self.file_path = file_path\n\ndef dicom2image(ele):\n    image_id = ele.file_path.split('/')[-1].split('.')[0]\n    dcm_file = pydicom.read_file(ele.file_path)\n    \n    PatientID = dcm_file.PatientID\n    series_id = dcm_file.SeriesInstanceUID\n    assert image_id == dcm_file.SOPInstanceUID\n    assert ele.StudyInstanceUID == dcm_file.StudyInstanceUID\n\n    data = apply_voi_lut(dcm_file.pixel_array, dcm_file)\n\n    if dcm_file.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n\n    image_path = './images/{}.png'.format(image_id)\n    cv2.imwrite(image_path, data)\n    return [PatientID, ele.StudyInstanceUID, series_id, image_id, dcm_file.SeriesNumber, dcm_file.InstanceNumber]\n\nsample_submission_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n\nmeles = []\nfor id in np.unique(sample_submission_df.id.values):\n    if '_study' not in id:\n        continue\n    StudyInstanceUID = id.replace('_study', '')\n\n    if StudyInstanceUID in public_test_1200_study:\n        continue\n\n    for rdir, _, files in os.walk('../input/siim-covid19-detection/test/{}'.format(StudyInstanceUID)):\n        for file in files:\n            file_path = os.path.join(rdir, file)\n            filename, file_extension = os.path.splitext(file_path)\n            if file_extension in ['.dcm', '.dicom']:\n                meles.append(ME(StudyInstanceUID, file_path))\n\np = Pool(4)\nresults = p.map(func=dicom2image, iterable = meles)\np.close()\ntest_df = pd.DataFrame(data=np.array(results), columns=['patientid', 'studyid', 'series_id', 'imageid', 'SeriesNumber', 'InstanceNumber'])\ntest_df.to_csv('./csv/test_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:55:48.610711Z","iopub.execute_input":"2021-07-31T12:55:48.612771Z","iopub.status.idle":"2021-07-31T12:55:52.421847Z","shell.execute_reply.started":"2021-07-31T12:55:48.612721Z","shell.execute_reply":"2021-07-31T12:55:52.420354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################################## crop lung area ##############################################\n### yolov5\n!cp -r ../input/siim-covid-src/detection_yolov5/* .\n!python predict_lung.py --ckpt_dir ../input/siim-covid-checkpoints/detection_yolov5_lung \\\n                        --output_dir ./det_predictions \\\n                        --output_file_name yolov5_lung_test_pred.pth \\\n                        --fold 3 \\\n                        --source ./images \\\n                        --img-size 512 \\\n                        --conf-thres 0.05 \\\n                        --iou-thres 0.5 \\\n                        --device 0\n!rm -rf ./utils ./models ./data *.py ./__pycache__","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:55:52.424041Z","iopub.execute_input":"2021-07-31T12:55:52.424424Z","iopub.status.idle":"2021-07-31T12:56:11.196426Z","shell.execute_reply.started":"2021-07-31T12:55:52.424373Z","shell.execute_reply":"2021-07-31T12:56:11.195393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################################## study level prediction ##############################################\n!cp -r ../input/siim-covid-src/classification_aux/* .\n!python predict_test.py --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/classification_aux_v4 \\\n                        --image_dir ./images \\\n                        --lung_pred_path ./det_predictions/yolov5_lung_test_pred.pth \\\n                        --output_dir ./cls_predictions \\\n                        --cfg ./configs/eb5_512_deeplabv3plus.yaml \\\n                        --folds 0 1 2 3 4 \\\n                        --num_tta 8 \\\n                        --batch-size 96 \\\n                        --workers 2\n                        \n!python predict_test.py --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/classification_aux_v4 \\\n                        --image_dir ./images \\\n                        --lung_pred_path ./det_predictions/yolov5_lung_test_pred.pth \\\n                        --output_dir ./cls_predictions \\\n                        --cfg ./configs/seresnet152d_512_unet.yaml \\\n                        --folds 0 1 2 3 4 \\\n                        --num_tta 8 \\\n                        --batch-size 96 \\\n                        --workers 2\n\n!python predict_test.py --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/classification_aux_v4 \\\n                        --image_dir ./images \\\n                        --lung_pred_path ./det_predictions/yolov5_lung_test_pred.pth \\\n                        --output_dir ./cls_predictions \\\n                        --cfg ./configs/eb6_448_linknet.yaml \\\n                        --folds 0 1 2 3 4 \\\n                        --num_tta 8 \\\n                        --batch-size 96 \\\n                        --workers 2\n\n!python predict_test.py --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/classification_aux_v4 \\\n                        --image_dir ./images \\\n                        --lung_pred_path ./det_predictions/yolov5_lung_test_pred.pth \\\n                        --output_dir ./cls_predictions \\\n                        --cfg ./configs/eb7_512_unetplusplus.yaml \\\n                        --folds 0 1 2 3 4 \\\n                        --num_tta 8 \\\n                        --batch-size 64 \\\n                        --workers 2\n\n!rm -rf *.py ./segmentation_models_pytorch ./configs ./__pycache__","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:56:11.197921Z","iopub.execute_input":"2021-07-31T12:56:11.198256Z","iopub.status.idle":"2021-07-31T12:58:51.471244Z","shell.execute_reply.started":"2021-07-31T12:56:11.198206Z","shell.execute_reply":"2021-07-31T12:58:51.470215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############################################## image level prediction ##############################################\n### yolov5\n!cp -r ../input/siim-covid-src/detection_yolov5/* .\n!python predict.py  --ckpt_dir ../input/siim-covid-checkpoints/detection_yolov5 \\\n                    --output_dir ./det_predictions \\\n                    --folds 0 1 2 3 4 \\\n                    --source ./images \\\n                    --img-size 768 \\\n                    --conf-thres 0.0005 \\\n                    --iou-thres 0.5 \\\n                    --device 0\n!rm -rf ./utils ./models ./data *.py ./__pycache__\n\n### faster rcnn\n!cp -r ../input/siim-covid-src/detection_fasterrcnn/* .\n!python predict_test.py --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/detection_fasterrcnn \\\n                        --image_dir ./images \\\n                        --output_dir ./det_predictions \\\n                        --cfg ./configs/resnet200d.yaml \\\n                        --folds 0 1 2 3 4 \\\n                        --batch-size 32 \\\n                        --workers 2\n!python predict_test.py --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/detection_fasterrcnn \\\n                        --image_dir ./images \\\n                        --output_dir ./det_predictions \\\n                        --cfg ./configs/resnet101d.yaml \\\n                        --folds 0 1 2 3 4 \\\n                        --batch-size 24 \\\n                        --workers 2\n!rm -rf *.py ./configs ./__pycache__\n\n### efficient det\n!cp -r ../input/siim-covid-src/detection_efficientdet/* .\n!python predict_test.py --model tf_efficientdet_d7 \\\n                        --amp --use-ema --num-classes 1 --native-amp -b 24 \\\n                        --output_dir ./det_predictions \\\n                        --test_df ./csv/test_df.csv \\\n                        --ckpt_dir ../input/siim-covid-checkpoints/detection_efficientdet \\\n                        --image_dir ./images \\\n                        --folds 0 1 2 3 4 \\\n                        --image-size 768\n!rm -rf ./effdet *.py ./__pycache__","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:58:51.475471Z","iopub.execute_input":"2021-07-31T12:58:51.475751Z","iopub.status.idle":"2021-07-31T13:01:59.225554Z","shell.execute_reply.started":"2021-07-31T12:58:51.475714Z","shell.execute_reply":"2021-07-31T13:01:59.224368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### remove temporary image dir\n!rm -rf ./images","metadata":{"execution":{"iopub.status.busy":"2021-07-31T13:01:59.227665Z","iopub.execute_input":"2021-07-31T13:01:59.228113Z","iopub.status.idle":"2021-07-31T13:01:59.904355Z","shell.execute_reply.started":"2021-07-31T13:01:59.228055Z","shell.execute_reply":"2021-07-31T13:01:59.903137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_study_output = []\nsubmission_image_output = []\n############################################## combine study + image level prediction ##############################################\ntest_df = pd.read_csv('./csv/test_df.csv')\n\neb5_study_pred = torch.load('./cls_predictions/timm-efficientnet-b5_512_deeplabv3plus_aux_fold0_1_2_3_4_test_pred.pth')['pred_dict']\neb6_study_pred = torch.load('./cls_predictions/timm-efficientnet-b6_448_linknet_aux_fold0_1_2_3_4_test_pred.pth')['pred_dict']\neb7_study_pred = torch.load('./cls_predictions/timm-efficientnet-b7_512_unetplusplus_aux_fold0_1_2_3_4_test_pred.pth')['pred_dict']\nsr152_study_pred = torch.load('./cls_predictions/timm-seresnet152d_320_512_unet_aux_fold0_1_2_3_4_test_pred.pth')['pred_dict']\n\nfor studyid, grp in test_df.groupby('studyid'):\n    preds = []\n    for _, row in grp.iterrows():\n        pred =  0.3*eb5_study_pred[row['imageid']] + \\\n                0.2*eb6_study_pred[row['imageid']] + \\\n                0.2*eb7_study_pred[row['imageid']] + \\\n                0.3*sr152_study_pred[row['imageid']]\n\n        preds.append(pred)\n        \n        boxes1, scores1, labels1, img_width, img_height = load_dict('./det_predictions/tf_efficientdet_d7_768_fold0_1_2_3_4_test_pred/{}.pkl'.format(row['imageid']))\n        \n        boxes2, scores2, labels2, img_width2, img_height2 = load_dict('./det_predictions/yolov5x6_768_fold0_1_2_3_4_test_pred/{}.pkl'.format(row['imageid']))\n        assert img_width2 == img_width and img_height2 == img_height\n\n        boxes3, scores3, labels3, img_width3, img_height3 = load_dict('./det_predictions/resnet200d_768_fold0_1_2_3_4_test_pred/{}.pkl'.format(row['imageid']))\n        assert img_width3 == img_width and img_height3 == img_height\n        \n        boxes4, scores4, labels4, img_width4, img_height4 = load_dict('./det_predictions/resnet101d_1024_fold0_1_2_3_4_test_pred/{}.pkl'.format(row['imageid']))\n        assert img_width4 == img_width and img_height4 == img_height\n        \n        boxes = boxes1 + boxes2 + boxes3 + boxes4\n        labels = labels1 + labels2 + labels3 + labels4\n\n        ### scale score of fasterrcnn to effdet and yolo score\n        scores3_tmp = []\n        for s in scores3:\n            tmp = [x*0.78 for x in s]\n            scores3_tmp.append(tmp)\n        scores3 = scores3_tmp\n        \n        scores4_tmp = []\n        for s in scores4:\n            tmp = [x*0.78 for x in s]\n            scores4_tmp.append(tmp)\n        scores4 = scores4_tmp\n\n        scores = scores1 + scores2 + scores3 + scores4\n\n        boxes, scores, labels = weighted_boxes_fusion(boxes, scores, labels, weights=None, iou_thr=0.6)\n        assert np.mean(labels) == 0\n        boxes = boxes.clip(0,1)\n\n        boxes[:,[0,2]] = boxes[:,[0,2]]*float(img_width)\n        boxes[:,[1,3]] = boxes[:,[1,3]]*float(img_height)\n        \n        neg_image_pred = 'none {} 0 0 1 1'.format(pred[0])\n        opacity_image_pred = []\n        for box, score in zip(boxes, scores):\n            opacity_image_pred.append('opacity {} {} {} {} {}'.format(score, box[0], box[1], box[2],box[3]))\n        image_pred = ' '.join([neg_image_pred] + opacity_image_pred)\n        submission_image_output.append(['{}_image'.format(row['imageid']), image_pred])\n\n    preds = np.array(preds)\n    preds = np.mean(preds, axis=0)\n\n    study_preds = []\n    for clsidx, clsname in enumerate(classes):\n        study_preds.append('{} {} 0 0 1 1'.format(study_submission_classes[clsname], preds[clsidx]))\n    study_preds = ' '.join(study_preds)\n    submission_study_output.append(['{}_study'.format(studyid), study_preds])\n\ndel eb5_study_pred\ndel eb6_study_pred\ndel eb7_study_pred\ndel sr152_study_pred\n\nsubmission_output = submission_study_output + submission_image_output\nsub_df = pd.DataFrame(data=np.array(submission_output), columns=['id','PredictionString'])\n\npublic_test_1200_submission_df = pd.read_csv('./submission_1200_study.csv')\n\nsub_df = pd.concat([public_test_1200_submission_df, sub_df], ignore_index=True)\nsub_df.to_csv('submission.csv', index=False)\nprint(sub_df.shape)\n\ndel submission_output\ndel submission_study_output\ndel submission_image_output\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T13:01:59.907964Z","iopub.execute_input":"2021-07-31T13:01:59.908259Z","iopub.status.idle":"2021-07-31T13:02:21.023613Z","shell.execute_reply.started":"2021-07-31T13:01:59.908224Z","shell.execute_reply":"2021-07-31T13:02:21.022835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### remove temporary prediction dir\n!rm -rf ./csv ./det_predictions ./cls_predictions ./submission_1200_study.csv","metadata":{"execution":{"iopub.status.busy":"2021-07-31T13:02:21.024845Z","iopub.execute_input":"2021-07-31T13:02:21.025185Z","iopub.status.idle":"2021-07-31T13:02:21.665968Z","shell.execute_reply.started":"2021-07-31T13:02:21.025147Z","shell.execute_reply":"2021-07-31T13:02:21.664981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T13:02:21.669007Z","iopub.execute_input":"2021-07-31T13:02:21.669267Z","iopub.status.idle":"2021-07-31T13:02:21.685129Z","shell.execute_reply.started":"2021-07-31T13:02:21.66924Z","shell.execute_reply":"2021-07-31T13:02:21.68427Z"},"trusted":true},"execution_count":null,"outputs":[]}]}