{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! python --version\n! ls -l /kaggle/input/covid19-detection-submissions\n! cp -r /kaggle/input/covid19-detection-submissions/package_freeze package_freeze\n\nimport glob, os, shutil\npkgs = glob.glob(\"package_freeze/*.xyz\")\n[shutil.move(p, p.replace(\".xyz\", \".tar.gz\")) for p in pkgs]\n\n# ! ls -l package_freeze\n# ! echo $PWD\n! pip install -q  --find-link /kaggle/working/package_freeze/ --no-index \"lightning-flash[image]\" \"python-gdcm\" pydicom \"opencv-python-headless\" pycocotools -U\n# ! pip install -q package_freeze/*whl package_freeze/*.tar.gz --no-index\n\n%reload_ext autoreload\n%autoreload 2\n\nimport gdcm\nimport pydicom\nimport flash","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:45:40.549856Z","iopub.execute_input":"2021-08-10T14:45:40.550251Z","iopub.status.idle":"2021-08-10T14:46:28.887298Z","shell.execute_reply.started":"2021-08-10T14:45:40.550162Z","shell.execute_reply":"2021-08-10T14:46:28.886025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing/convert images","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\nfrom copy import deepcopy\nfrom pydicom.pixel_data_handlers import apply_voi_lut\n\n\ndef load_image(path_file: str, meta: dict, spacing: float = 1.0, percentile: bool = True):\n    dicom = pydicom.dcmread(path_file)\n    try:\n        img = apply_voi_lut(dicom.pixel_array, dicom)\n    except RuntimeError as err:\n        return None, meta\n    if dicom.PhotometricInterpretation == 'MONOCHROME1':\n        img = img.max() - img\n    p_low = np.percentile(img, 1) if percentile else img.min()\n    p_high = np.percentile(img, 99) if percentile else img.max()\n    # normalize\n    img = (img.astype(float) - p_low) / (p_high - p_low)\n    meta.update({\n        'original_spacing': dicom.ImagerPixelSpacing,\n        'original_image_shape': img.shape,\n        'spacing': dicom.ImagerPixelSpacing,\n        'image_shape': img.shape,\n    })\n    if spacing:\n        factor = np.array(meta['spacing']) / spacing\n        dims = tuple((np.array(img.shape[::-1]) * factor).astype(int))\n        img = cv2.resize(img, dsize=dims, interpolation=cv2.INTER_LINEAR)\n        meta.update({'spacing': (spacing, spacing), 'image_shape': img.shape})\n    return img, meta","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-10T14:46:28.891192Z","iopub.execute_input":"2021-08-10T14:46:28.891503Z","iopub.status.idle":"2021-08-10T14:46:29.102059Z","shell.execute_reply.started":"2021-08-10T14:46:28.891471Z","shell.execute_reply":"2021-08-10T14:46:29.101068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm\n\nSPACING = 0.75\nBASE_PATH = '/kaggle/input/siim-covid19-detection'\nPATH_OUT = f\"/home/jovyan/work/coco-dataset_spacing-{SPACING}\"\nPATH_OUT_IMAGE = os.path.join(PATH_OUT, \"images\")\nPATH_OUT_LABEL = os.path.join(PATH_OUT, \"labels\")\n\nfor d in (PATH_OUT_IMAGE, PATH_OUT_LABEL):\n    os.makedirs(d, exist_ok=True)\n    for dd in (\"train\", \"test\"):\n        os.makedirs(os.path.join(d, dd), exist_ok=True)\n\n\ndef conver_image(id_row, dir_name):\n    _, row = id_row\n    # phase = \"train\" if np.random.random() < 0.8 else \"valid\"\n    img, meta = load_image(os.path.join(BASE_PATH, row['path']), dict(row), spacing=SPACING)\n    plt.imsave(os.path.join(PATH_OUT_IMAGE, dir_name, f\"{row['name']}.jpg\"), img, cmap='gray')\n    return meta","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:46:29.104079Z","iopub.execute_input":"2021-08-10T14:46:29.104589Z","iopub.status.idle":"2021-08-10T14:46:29.180825Z","shell.execute_reply.started":"2021-08-10T14:46:29.104534Z","shell.execute_reply":"2021-08-10T14:46:29.179557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfrom tqdm.autonotebook import tqdm\nfrom multiprocessing import Pool\nfrom functools import partial\n\nfound_images = glob.glob(os.path.join(BASE_PATH, 'test', '*', '*', '*.dcm'))\n\ntest_images = pd.DataFrame([{\n    \"name\": os.path.splitext(os.path.basename(p))[0],\n    \"path\": os.path.sep.join(p.split(os.path.sep)[-4:]),\n    \"study\": p.split(os.path.sep)[-3],\n} for p in found_images])\ndisplay(test_images.head())\n\n\npool = Pool(os.cpu_count())\ntest_images = list(pool.imap_unordered(partial(conver_image, dir_name=\"test\"), tqdm(test_images.iterrows(), total=len(test_images))))\npool.close()\npool.join()\n\ntest_images = pd.DataFrame(test_images)\ndisplay(test_images.head())","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:46:29.182901Z","iopub.execute_input":"2021-08-10T14:46:29.183389Z","iopub.status.idle":"2021-08-10T14:50:21.985634Z","shell.execute_reply.started":"2021-08-10T14:46:29.183334Z","shell.execute_reply":"2021-08-10T14:50:21.984522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predictions","metadata":{}},{"cell_type":"code","source":"! mkdir -p ~/.icevision/fonts/\n! cp /kaggle/input/covid19-detection-submissions/SpaceGrotesk-Medium.ttf ~/.icevision/fonts/SpaceGrotesk-Medium.ttf","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:50:21.987375Z","iopub.execute_input":"2021-08-10T14:50:21.987912Z","iopub.status.idle":"2021-08-10T14:50:23.499942Z","shell.execute_reply.started":"2021-08-10T14:50:21.987865Z","shell.execute_reply":"2021-08-10T14:50:23.498687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from flash.image import ObjectDetectionData, ObjectDetector\nfrom tqdm.autonotebook import tqdm\nfrom flash.image.detection.data import ObjectDetectionPreprocess\n\nmodel = ObjectDetector.load_from_checkpoint(\"/kaggle/input/covid19-detection-submissions/object_detection_model.pt\", pretrained=False)\nmodel._preprocess = ObjectDetectionPreprocess(image_size=512)\ndisplay(test_images.head())","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:50:23.501986Z","iopub.execute_input":"2021-08-10T14:50:23.502426Z","iopub.status.idle":"2021-08-10T14:50:53.918691Z","shell.execute_reply.started":"2021-08-10T14:50:23.502382Z","shell.execute_reply":"2021-08-10T14:50:53.917857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. Detect objects in a few images!\npredictions = []\nmodel.to(\"cuda\")\nfor _, row in tqdm(test_images.iterrows(), total=len(test_images)):\n    p_img = os.path.join(PATH_OUT_IMAGE, \"test\", f\"{row['name']}.jpg\")\n    preds = model.predict([p_img])\n    rec = {**dict(row), \"predictions\": preds[0]}\n    predictions.append(rec)","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:50:53.919975Z","iopub.execute_input":"2021-08-10T14:50:53.920336Z","iopub.status.idle":"2021-08-10T14:54:04.805595Z","shell.execute_reply.started":"2021-08-10T14:50:53.920300Z","shell.execute_reply":"2021-08-10T14:54:04.804590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Post-processing","metadata":{}},{"cell_type":"code","source":"submission = []\nfor rec in predictions:\n    # print(rec[\"predictions\"].as_dict()[\"detection\"])\n    fact = rec[\"spacing\"][0] / rec[\"original_spacing\"][0]\n    lbs = set(rec[\"predictions\"].as_dict()[\"detection\"][\"label_ids\"])\n    bboxes = [] if lbs else [\"none 1 0 0 1 1\"]\n    det = rec[\"predictions\"].as_dict()[\"detection\"]\n    for bb, score in zip(det[\"bboxes\"], det[\"scores\"]):\n        bboxes.append(f\"opacity {score} {bb.xmin * fact} {bb.ymin * fact} {bb.xmax * fact} {bb.ymax * fact}\")\n    submission.append({\n        \"Id\": f\"{rec['name']}_image\",\n        \"PredictionString\": \" \".join(bboxes),\n    })\n\nLABELS = (\"negative\", \"typical\", \"indeterminate\", \"atypical\")\ndf_predictions = pd.DataFrame(predictions)\nfor n, dfg in df_predictions.groupby(\"study\"):\n    # print(dfg[\"predictions\"].apply(lambda p: p.as_dict()))\n    lbs = []\n    for _, row in dfg.iterrows():\n        lbs += row[\"predictions\"].as_dict()[\"detection\"][\"label_ids\"]\n    if not lbs:\n        pred = \"negative 1 0 0 1 1\"\n    else:\n        pred = \" \".join([f\"{LABELS[lb]} 1 0 0 1 1\" for lb in set(lbs)])\n    submission.append({\n        \"Id\": f\"{n}_study\",\n        \"PredictionString\": pred,\n    })\n\nsubmission = pd.DataFrame(submission).drop_duplicates(subset=['Id'])\ndisplay(submission.head())\nsubmission.to_csv(\"submission.csv\", index=None)","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:54:04.807920Z","iopub.execute_input":"2021-08-10T14:54:04.808297Z","iopub.status.idle":"2021-08-10T14:54:05.153577Z","shell.execute_reply.started":"2021-08-10T14:54:04.808258Z","shell.execute_reply":"2021-08-10T14:54:05.152779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! cat submission.csv","metadata":{"execution":{"iopub.status.busy":"2021-08-10T14:54:05.154948Z","iopub.execute_input":"2021-08-10T14:54:05.155329Z","iopub.status.idle":"2021-08-10T14:54:05.939042Z","shell.execute_reply.started":"2021-08-10T14:54:05.155293Z","shell.execute_reply":"2021-08-10T14:54:05.938179Z"},"trusted":true},"execution_count":null,"outputs":[]}]}