{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"text-align: center; font-family: Verdana; font-size: 32px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; font-variant: small-caps; letter-spacing: 3px; color: #186b75; background-color: #ffffff;\">SIIM – COVID19 Detection<br><br><font color=\"red\">EfficientNetV2-B3</font></h1>\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\">Study Level Image Exploration</h2>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5><br>\n\n<br>\n\n<center>💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠</center>\n<center>🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫</center>\n<center>💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠</center>\n<center>🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪🔬🩺💊🧬🦠🧫🧪</center>\n<center>🩺🩺🩺🩺🦠🦠🦠🩺🩺🩺🩺</center>\n<center>🩺🩺🩺</center>\n\n<br>\n","metadata":{"papermill":{"duration":0.044805,"end_time":"2021-02-05T02:17:44.538251","exception":false,"start_time":"2021-02-05T02:17:44.493446","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<h2 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: darkred; background-color: #ffffff;\">TABLE OF CONTENTS</h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#imports\">0&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#background_information\">1&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#setup\">2&nbsp;&nbsp;&nbsp;&nbsp;SETUP</a></h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#helper_functions\">3&nbsp;&nbsp;&nbsp;&nbsp;HELPER FUNCTIONS</a></h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#tabular_data\">4&nbsp;&nbsp;&nbsp;&nbsp;TABULAR DATA</a></h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#image_data\">5&nbsp;&nbsp;&nbsp;&nbsp;IMAGE DATA</a></h2>\n\n---\n\n<h2 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\"><a href=\"#find_duplicates\">6&nbsp;&nbsp;&nbsp;&nbsp;IDENTIFY DUPLICATES</a></h2>\n\n---\n\n<br>","metadata":{"papermill":{"duration":0.043006,"end_time":"2021-02-05T02:17:44.624815","exception":false,"start_time":"2021-02-05T02:17:44.581809","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: darkred;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS</h1>","metadata":{"papermill":{"duration":0.04267,"end_time":"2021-02-05T02:17:44.71043","exception":false,"start_time":"2021-02-05T02:17:44.66776","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Installs\n!cp /kaggle/input/gdcm-conda-install/gdcm.tar .\n!tar -xvzf gdcm.tar\n!conda install --offline ./gdcm/gdcm-2.8.9-py37h71b2a6d_0.tar.bz2\n!rm -rf ./gdcm.tar\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\n\nimport pydicom\nfrom pydicom import dcmread\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport imageio\nimport IPython\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nimport plotly\nimport PIL\nimport cv2\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")\n\n# To give access to automl files\nsys.path.append(\"/kaggle/input/automl-efficientdet-efficientnetv2/automl\")\nsys.path.append(\"/kaggle/input/automl-efficientdet-efficientnetv2/automl/brain_automl\")\nsys.path.append(\"/kaggle/input/automl-efficientdet-efficientnetv2/automl/brain_automl/efficientdet\")\nsys.path.append(\"/kaggle/input/automl-efficientdet-efficientnetv2/automl/brain_automl/efficientnetv2\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":41.017298,"end_time":"2021-02-05T02:18:25.770952","exception":false,"start_time":"2021-02-05T02:17:44.753654","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-06-05T15:05:28.241585Z","iopub.execute_input":"2021-06-05T15:05:28.241916Z","iopub.status.idle":"2021-06-05T15:05:41.148566Z","shell.execute_reply.started":"2021-06-05T15:05:28.241885Z","shell.execute_reply":"2021-06-05T15:05:41.147574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: darkred; background-color: #ffffff;\" id=\"background_information\">1&nbsp;&nbsp;BACKGROUND INFORMATION</h1>","metadata":{"papermill":{"duration":0.045609,"end_time":"2021-02-05T02:18:25.862323","exception":false,"start_time":"2021-02-05T02:18:25.816714","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"**<br>\n\n<h2 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\">1.1  THE DATA</h2>\n\n---\n\n<b style=\"text-decoration: underline; font-family: Verdana;\">BACKGROUND INFORMATION</b>\n\nIn this competition, we are identifying and localizing COVID-19 abnormalities on chest radiographs. <br>**This is an object detection and classification problem.**\n\nFor each test image, you will be predicting a bounding box and class for all findings. \n* If you predict that there are no findings, you should create a prediction of **`\"none 1 0 0 1 1\"`** \n    * \"none\" is the class ID for no finding, and this provides a one-pixel bounding box with a confidence of 1.0\n\nFurther, for each test study, you should make a determination within the following labels:\n\n> **`'Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance'`**\n\nTo make a prediction of one of the above labels, create a prediction string similar to the \"none\" class above: \n* i.e. **`atypical 1 0 0 1 1`**\n\n---\n\n**MESSAGE FROM THE COMPETITION HOST ON LABEL AND BBOX DETAILS:**\n\nIn this challenge, the chest radiographs (CXRs) were categorized using a specific grading schema, based on a published paper:\n\n[**Litmanovich DE, Chung M, Kirkbride RR, Kicska G, Kanne JP. Review of chest radiograph findings of COVID-19 pneumonia and suggested reporting language. Journal of thoracic imaging. 2020 Nov 14;35(6):354-60.**](https://journals.lww.com/thoracicimaging/Fulltext/2020/11000/Review_of_Chest_Radiograph_Findings_of_COVID_19.4.aspx)\n\nPer the grading schema, chest radiographs are classified into one of four categories, which are mutually exclusive:\n\n1. **Typical Appearance**: Multifocal bilateral, peripheral opacities with rounded morphology, lower lung–predominant distribution\n2. **Indeterminate Appearance**: Absence of typical findings AND unilateral, central or upper lung predominant distribution\n3. **Atypical Appearance**: Pneumothorax, pleural effusion, pulmonary edema, lobar consolidation, solitary lung nodule or mass, diffuse tiny nodules, cavity\n4. **Negative for Pneumonia**: No lung opacities\n\nBounding boxes were placed on lung opacities, whether typical or indeterminate. Bounding boxes were also placed on some atypical findings including solitary lobar consolidation, nodules/masses, and cavities. Bounding boxes were not placed on pleural effusions, or pneumothoraces. No bounding boxes were placed for the negative for pneumonia category.\n\nIn cases of multiple adjacent opacities, we opted for one large bounding box, rather than multiple adjacent smaller boxes, to improve consistency in the labeling.\n\nAnnotators did have access to the COVID status for each patient, but were asked to adhere to the grading system above irrespective of the status. As such, some patients who were COVID negative still had chest radiographs with typical appearances. Similarly, some patients who were COVID positive had atypical appearances, or were negative for pneumonia (no lung opacities), because the grading system is based off the chest radiographic findings alone.\n\nThe goal in this challenge is to determine the appropriate category for each radiograph, as well as localize the lung opacities with a bounding box prediction.\n\n---\n\nThe images are in DICOM format, which means they contain additional data that might be useful for visualizing and classifying.\nNote that the images are in **DICOM** format, which means they contain additional data that might be useful for visualizing and classifying.\n\n![Example Radiographs](https://i.imgur.com/QWmbhXx.png)\n\n<br>\n\n<b style=\"text-decoration: underline; font-family: Verdana;\">DATASET INFORMATION</b>\n\nThe **train dataset** comprises **`6,334`** chest scans in **DICOM** format, which were de-identified to protect patient privacy. \n\nNote that all images are stored in paths with the form **`study/series/image`**. \n* The **`study`** ID here relates directly to the study-level predictions\n* the **`image`** ID is the ID used for image-level predictions\n\nThe **test dataset** is of roughly the same scale as the training dataset. \n* As this is a kernels only competition we shsould plan accordingly\n* i.e. we should be able to infer on the entirety of the training dataset within the submission kernel\n\n<br>\n\n<b style=\"text-decoration: underline; font-family: Verdana;\">DATA FILES</b>\n> **`train_study_level.csv`**\n> * **`id`** - unique study identifier\n> * **Negative for Pneumonia** - **`1`** if the study is negative for pneumonia, **`0`** otherwise\n> * **Typical Appearance** - **`1`** if the study has this appearance, **`0`** otherwise\n> * **Indeterminate Appearance**  - **`1`** if the study has this appearance, **`0`** otherwise\n> * **Atypical Appearance**  - **`1`** if the study has this appearance, **`0`** otherwise\n\n> **`train_image_level.csv`**\n> * **`id`** - unique image identifier\n> * **`boxes`** - bounding boxes in easily-readable dictionary format\n> * **`label`** - the correct prediction label for the provided bounding boxes","metadata":{"papermill":{"duration":0.04413,"end_time":"2021-02-05T02:18:25.953034","exception":false,"start_time":"2021-02-05T02:18:25.908904","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<h2 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\">1.2  THE GOAL</h2>\n\n---\n\nIn this competition, you’ll identify and localize COVID-19 abnormalities on chest radiographs. In particular, you'll categorize the radiographs as one of a possible **`4`** categories. \n\nIn this competition, we are making predictions at both a study (multi-image) and image level.\n* **`negative for pneumonia`** or **`typical`**, **`indeterminate`**, or **`atypical`** \n     \nYou'll work with a dataset consisting of **`8,781`** scans that have been annotated by experienced radiologists. You can train your model with **`6,334`** independently-labeled images and you will be evaluated on a test set of **`2,447`** images. \n\nThe challenge uses the standard PASCAL VOC 2010 mean Average Precision (mAP) at IoU > 0.5.\n* Note that the linked document describes VOC 2012, which differs in some minor ways (e.g. there is no concept of \"difficult\" classes in VOC 2010). The P/R curve and AP calculations remain the same.\n\n<br>\n\n<center>🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺</center>\n\n<center><font color=\"red\"><b>If successful, you'll help radiologists diagnose the millions of COVID-19 patients more confidently and quickly.</b></font></center>\n\n<center>🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺🩺</center>","metadata":{"papermill":{"duration":0.044853,"end_time":"2021-02-05T02:18:26.044977","exception":false,"start_time":"2021-02-05T02:18:26.000124","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<h2 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: darkred; background-color: #ffffff;\">1.3  ADDITIONAL INFORMATION ON ABNORMALITIES</h2>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">Negative for Pneumonia</b>\n* No lung opacities\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">Typical Appearance</b>\n* Multifocal bilateral, peripheral opacities with rounded morphology, lower lung–predominant distribution\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">Indeterminate Appearance</b>\n* Absence of typical findings AND unilateral, central or upper lung predominant distribution\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">Atypical Appearance</b>\n* Pneumothorax, pleural effusion, pulmonary edema, lobar consolidation, solitary lung nodule or mass, diffuse tiny nodules, cavity","metadata":{"papermill":{"duration":0.046066,"end_time":"2021-02-05T02:18:26.137054","exception":false,"start_time":"2021-02-05T02:18:26.090988","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"setup\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: darkred; background-color: #ffffff;\" id=\"setup\">2&nbsp;&nbsp;NOTEBOOK SETUP</h1>","metadata":{"papermill":{"duration":0.045324,"end_time":"2021-02-05T02:18:26.226344","exception":false,"start_time":"2021-02-05T02:18:26.18102","status":"completed"},"tags":[]}},{"cell_type":"code","source":"TRAIN_CSV_PATH = \"/kaggle/input/siim-covid19-updated-train-labels/updated_train_labels.csv\"\nSS_CSV_PATH = \"/kaggle/input/siim-covid19-updated-train-labels/updated_sample_submission.csv\"\n\nprint(\"\\n\\nCOMBINED AND EXPLODED TRAIN DATAFRAME\\n\\n\")\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ndisplay(train_df)\n\nprint(\"\\n\\nSAMPLE SUBMISSION DATAFRAME\\n\\n\")\nss_df = pd.read_csv(SS_CSV_PATH)\ndisplay(ss_df)\n\nimage_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\nall_image_ids = image_df.id.str.replace(\"_image\", \"\")\nbbox_image_ids = image_df.dropna().id.str.replace(\"_image\", \"\")","metadata":{"papermill":{"duration":0.717244,"end_time":"2021-02-05T02:18:26.987446","exception":false,"start_time":"2021-02-05T02:18:26.270202","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-06-05T12:25:03.913782Z","iopub.execute_input":"2021-06-05T12:25:03.91418Z","iopub.status.idle":"2021-06-05T12:25:04.165196Z","shell.execute_reply.started":"2021-06-05T12:25:03.914139Z","shell.execute_reply":"2021-06-05T12:25:04.164281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"helper_functions\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: darkred; background-color: #ffffff;\" id=\"helper_functions\">3&nbsp;&nbsp;HELPER FUNCTIONS</h1>","metadata":{"papermill":{"duration":0.045205,"end_time":"2021-02-05T02:18:27.079444","exception":false,"start_time":"2021-02-05T02:18:27.034239","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    \"\"\" Convert dicom file to numpy array \n    \n    Args:\n        path (str): Path to the dicom file to be converted\n        voi_lut (bool): Whether or not VOI LUT is available\n        fix_monochrome (bool): Whether or not to apply monochrome fix\n        \n    Returns:\n        Numpy array of the respective dicom file \n        \n    \"\"\"\n    # Use the pydicom library to read the dicom file\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to \n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n        \n    # The XRAY may look inverted\n    #   - If we want to fix this we can\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    \n    # Normalize the image array and return\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\n\ndef dicom2array_2(fname, target_size=512, use_clahe=True, clip_limit=2., grid_size=(8,8)):\n    dicom = pydicom.dcmread(fname)\n    data = apply_voi_lut(dicom.pixel_array, dicom)\n    im = data - np.min(data)\n    im = 255. * im / np.max(im)\n    \n    if dicom.PhotometricInterpretation == \"MONOCHROME1\": # check for inverted image\n        im = 255. - im\n    \n    if use_clahe:\n        clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=grid_size)\n        climg = clahe.apply(im.astype('uint8'))\n        img = Image.fromarray(climg.astype('uint8'), 'L')\n    else:\n        img = Image.fromarray(im.astype('uint8'), 'L')\n    org_size = img.size\n    \n    if max(img.size) > target_size:\n        img.thumbnail((target_size, target_size), Image.ANTIALIAS)\n    \n    return np.asarray(img)\n\ndef get_absolute_file_paths(directory):\n    all_abs_file_paths = []\n    for dirpath,_,filenames in os.walk(directory):\n        for f in filenames:\n            all_abs_file_paths.append(os.path.abspath(os.path.join(dirpath, f)))\n    print(all_abs_file_paths)        \n    return all_abs_file_paths","metadata":{"execution":{"iopub.status.busy":"2021-06-05T12:25:11.01467Z","iopub.execute_input":"2021-06-05T12:25:11.014998Z","iopub.status.idle":"2021-06-05T12:25:11.026535Z","shell.execute_reply.started":"2021-06-05T12:25:11.014971Z","shell.execute_reply":"2021-06-05T12:25:11.025701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    display(study_df)\nexcept:\n    study_df = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")\n    study_df = study_df[study_df.id.str.contains(\"study\")]\n    study_df[\"id\"] = study_df[\"id\"].str.replace(\"_study\", \"\")\n    study_df[\"study_dir\"] = \"/kaggle/input/siim-covid19-detection/train/\"+study_df[\"id\"]\n    study_df[\"images_per_study\"] = study_df.study_dir.progress_apply(lambda x: len(get_absolute_file_paths(x)))\nmultiple_images_per_study_df = study_df[study_df.images_per_study>1].reset_index(drop=True)\n\nfor dir_path in multiple_images_per_study_df.study_dir.values:\n    image_paths = get_absolute_file_paths(dir_path)\n    if len(image_paths)<=4:\n        plt.figure(figsize=(18,4))\n        plt.suptitle(f\"\\n\\nSTUDY: {dir_path.rsplit('/', 1)[1]}\", fontsize=16, fontweight=\"bold\")\n        for j, x in enumerate(image_paths):\n            if any(True for xx in all_image_ids if xx in x):\n                title = \"\\nINCLUDED IN IMAGE LEVEL\"\n                if any(True for xx in bbox_image_ids if xx in x):\n                    title += \" - W/ BBOX!!!\"\n            else:\n                title = \"xxxxxxxxxxxxxxxxxxxxxxx\"\n            plt.subplot(1,4,j+1)\n            plt.imshow(dicom2array_2(x))\n            plt.axis(False)\n            plt.title(title, fontweight=\"bold\")\n    elif len(image_paths)<=8:\n        plt.figure(figsize=(18,8))\n        plt.suptitle(f\"\\n\\nSTUDY: {dir_path.rsplit('/', 1)[1]}\", fontsize=16, fontweight=\"bold\")\n        for j, x in enumerate(image_paths):\n            if any(True for xx in all_image_ids if xx in x):\n                title = \"\\nINCLUDED IN IMAGE LEVEL\"\n                if any(True for xx in bbox_image_ids if xx in x):\n                    title += \" - W/ BBOX!!!\"\n            else:\n                title = \"xxxxxxxxxxxxxxxxxxxxxxx\"\n            plt.subplot(2,4,j+1)\n            plt.imshow(dicom2array_2(x))\n            plt.axis(False)\n            plt.title(title, fontweight=\"bold\")\n    else:\n        plt.figure(figsize=(18,12))\n        plt.suptitle(f\"\\n\\nSTUDY: {dir_path.rsplit('/', 1)[1]}\", fontsize=16, fontweight=\"bold\")\n        for j, x in enumerate(image_paths):\n            if any(True for xx in all_image_ids if xx in x):\n                title = \"\\nINCLUDED IN IMAGE LEVEL\"\n                if any(True for xx in bbox_image_ids if xx in x):\n                    title += \" - W/ BBOX!!!\"\n            else:\n                title = \"xxxxxxxxxxxxxxxxxxxxxxx\"\n            plt.subplot(3,4,j+1)\n            plt.imshow(dicom2array_2(x))\n            plt.title(title, fontweight=\"bold\")\n            plt.axis(False)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","papermill":{"duration":0.083595,"end_time":"2021-02-05T02:18:27.209879","exception":false,"start_time":"2021-02-05T02:18:27.126284","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-06-04T13:20:56.344007Z","iopub.execute_input":"2021-06-04T13:20:56.344341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nimport sys\nimport random\nimport math\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport json\nimport pydicom\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport pandas as pd \nimport glob ","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","papermill":{"duration":0.083595,"end_time":"2021-02-05T02:18:27.209879","exception":false,"start_time":"2021-02-05T02:18:27.126284","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-06-05T15:06:00.397164Z","iopub.execute_input":"2021-06-05T15:06:00.397518Z","iopub.status.idle":"2021-06-05T15:06:00.741801Z","shell.execute_reply.started":"2021-06-05T15:06:00.397483Z","shell.execute_reply":"2021-06-05T15:06:00.74095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input'\n\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:06:26.794461Z","iopub.execute_input":"2021-06-05T15:06:26.794787Z","iopub.status.idle":"2021-06-05T15:06:26.798722Z","shell.execute_reply.started":"2021-06-05T15:06:26.794757Z","shell.execute_reply":"2021-06-05T15:06:26.797836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\nos.chdir('Mask_RCNN')\n#!python setup.py -q install","metadata":{"execution":{"iopub.status.busy":"2021-06-05T13:15:24.03565Z","iopub.execute_input":"2021-06-05T13:15:24.035982Z","iopub.status.idle":"2021-06-05T13:15:24.692664Z","shell.execute_reply.started":"2021-06-05T13:15:24.035947Z","shell.execute_reply":"2021-06-05T13:15:24.691598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Mask RCNN\nsys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:06:31.405293Z","iopub.execute_input":"2021-06-05T15:06:31.405737Z","iopub.status.idle":"2021-06-05T15:06:31.458527Z","shell.execute_reply.started":"2021-06-05T15:06:31.405697Z","shell.execute_reply":"2021-06-05T15:06:31.457531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dicom_dir = os.path.join(DATA_DIR, 'train')\ntest_dicom_dir = os.path.join(DATA_DIR, 'test')","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:06:36.585949Z","iopub.execute_input":"2021-06-05T15:06:36.586284Z","iopub.status.idle":"2021-06-05T15:06:36.59059Z","shell.execute_reply.started":"2021-06-05T15:06:36.586251Z","shell.execute_reply":"2021-06-05T15:06:36.589574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.dcm')\n    return list(set(dicom_fps))\n\ndef parse_dataset(dicom_dir, anns): \n    image_fps = get_dicom_fps(dicom_dir)\n    ##print(image_fps)\n    image_annotations = {fp: [] for fp in image_fps}\n    ##print(image_annotations)\n    for index, row in anns.iterrows(): \n        fp = row['dcm_path']\n        image_annotations[fp].append(row)\n    return image_fps, image_annotations ","metadata":{"execution":{"iopub.status.busy":"2021-06-05T13:15:38.570513Z","iopub.execute_input":"2021-06-05T13:15:38.570844Z","iopub.status.idle":"2021-06-05T13:15:38.577073Z","shell.execute_reply.started":"2021-06-05T13:15:38.570807Z","shell.execute_reply":"2021-06-05T13:15:38.575936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The following parameters have been selected to reduce running time for demonstration purposes \n# These are not optimal \n\nclass DetectorConfig(Config):\n    \"\"\"Configuration for training pneumonia detection on the RSNA pneumonia dataset.\n    Overrides values in the base Config class.\n    \"\"\"\n    \n    # Give the configuration a recognizable name  \n    NAME = 'covid'\n    \n    # Train on 1 GPU and 8 images per GPU. We can put multiple images on each\n    # GPU because the images are small. Batch size is 8 (GPUs * images/GPU).\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 8 \n    \n    BACKBONE = 'resnet50'\n    \n    NUM_CLASSES = 1+4  # background + 1 pneumonia classes\n    \n    IMAGE_MIN_DIM = 256\n    IMAGE_MAX_DIM = 256\n    RPN_ANCHOR_SCALES = (32, 64, 128, 256)\n    TRAIN_ROIS_PER_IMAGE = 32\n    MAX_GT_INSTANCES = 3\n    DETECTION_MAX_INSTANCES = 3\n    DETECTION_MIN_CONFIDENCE = 0.9\n    DETECTION_NMS_THRESHOLD = 0.1\n\n    STEPS_PER_EPOCH = 100\n    \nconfig = DetectorConfig()\nprint(config)\nconfig.display()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:06:48.636984Z","iopub.execute_input":"2021-06-05T15:06:48.637304Z","iopub.status.idle":"2021-06-05T15:06:48.652377Z","shell.execute_reply.started":"2021-06-05T15:06:48.637275Z","shell.execute_reply":"2021-06-05T15:06:48.651369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import skimage\nimport numpy\nclass DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, image_fps, image_annotations,orig_height , orig_width):\n        super().__init__(self)\n        \n        # Add classes\n        self.add_class('covid', 1, 'Negative for Pneumonia')\n        self.add_class('covid', 2, 'Typical Appearance')\n        self.add_class('covid', 3, 'Indeterminate Appearance')\n        self.add_class('covid', 4, 'Atypical Appearance')\n        \n   \n        # add images \n        for i, fp in enumerate(image_fps):\n            annotations = image_annotations[fp]\n            self.add_image('covid', image_id=i, path=fp, \n                           annotations=annotations, orig_height=orig_height, orig_width=orig_width)\n            \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        fp = info['path']\n        image = cv2.imread(fp)\n        plt.imshow(image)\n        plt.show()\n        data = np.asarray( image, dtype='uint8' )\n        \n        # If grayscale. Convert to RGB for consistency.\n        if len(data.shape) != 3 or data.shape[2] != 3:\n             data = np.stack((data,) * 3, -1)\n        return data\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        annotations = info['annotations']\n        count = len(annotations)\n        image = dataset_train.load_image(image_id)\n        height, width = image.shape[:2]\n        if count == 0:\n            mask = np.zeros((info['orig_height'],info['orig_width'] , 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            mask = np.zeros((info['orig_height'],info['orig_width'] , 1), dtype=np.uint8)\n            class_ids = np.zeros((count,), dtype=np.int32)\n            for i, a in enumerate(annotations):\n                    print(i)\n                    x = int(a['xmin'])\n                    y = int(a['ymin'])\n                    w = int(a['xmax'])\n                    h = int(a['ymax'])\n                    mask_instance = mask[:, :, i].copy()\n                    cv2.rectangle(mask_instance, (x, y), (w, h), 255, -1)\n                    mask[:, :, i] = mask_instance\n                    class_ids[i] = i\n        return mask.astype(np.bool), class_ids.astype(np.int32)\n ","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:37:28.136198Z","iopub.execute_input":"2021-06-07T20:37:28.136546Z","iopub.status.idle":"2021-06-07T20:37:28.155266Z","shell.execute_reply.started":"2021-06-07T20:37:28.136517Z","shell.execute_reply":"2021-06-07T20:37:28.154263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset\nanns = pd.read_csv(os.path.join(DATA_DIR, '../input/siim-covid19-updated-train-labels/updated_train_labels.csv'))\nanns.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:07:01.923336Z","iopub.execute_input":"2021-06-05T15:07:01.923705Z","iopub.status.idle":"2021-06-05T15:07:02.015437Z","shell.execute_reply.started":"2021-06-05T15:07:01.923673Z","shell.execute_reply":"2021-06-05T15:07:02.014472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##image_fps, image_annotations = parse_dataset(train_dicom_dir, anns=anns)\ndef get_absolute_file_paths(directory):\n    all_abs_file_paths = []\n    for dirpath,_,filenames in os.walk(directory):\n        for f in filenames:\n            all_abs_file_paths.append(os.path.abspath(os.path.join(dirpath, f)))\n    return all_abs_file_paths\n##image_fps[]\n\nimage_fps=anns.dcm_path.tolist()\n##print(image_fps)\n\n\nds = pydicom.read_file(image_fps[0])         # read dicom image from filepath \nimage = ds.pixel_array     \nds\n\n# Original DICOM image size: 1024 x 1024\nORIG_SIZE = 2048\n\n######################################################################\n# Modify this line to use more or fewer images for training/validation. \n# To use all images, do: image_fps_list = list(image_fps)\nimage_fps_list = list(image_fps[:1000]) \n#####################################################################\n\n# split dataset into training vs. validation dataset \n# split ratio is set to 0.9 vs. 0.1 (train vs. validation, respectively)\nsorted(image_fps_list)\nrandom.seed(42)\nrandom.shuffle(image_fps_list)\n\nvalidation_split = 0.1\nsplit_index = int((1 - validation_split) * len(image_fps_list))\n\nimage_fps_train = image_fps_list[:split_index]\nimage_fps_val = image_fps_list[split_index:]\n\nprint(len(image_fps_train), len(image_fps_val))\n\n","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:07:16.224061Z","iopub.execute_input":"2021-06-05T15:07:16.224394Z","iopub.status.idle":"2021-06-05T15:07:16.268507Z","shell.execute_reply.started":"2021-06-05T15:07:16.224355Z","shell.execute_reply":"2021-06-05T15:07:16.26763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_annotations = {fp: [] for fp in image_fps}\nfor index, row in anns.iterrows():\n    image_annotations[row[1]].append(row)\n##print(image_annotations)        \n       \n   \n\n     ","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:07:23.463979Z","iopub.execute_input":"2021-06-05T15:07:23.464305Z","iopub.status.idle":"2021-06-05T15:07:31.380883Z","shell.execute_reply.started":"2021-06-05T15:07:23.464273Z","shell.execute_reply":"2021-06-05T15:07:31.379738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the training dataset\ndataset_train = DetectorDataset(image_fps_train, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:16:00.823343Z","iopub.execute_input":"2021-06-05T15:16:00.823807Z","iopub.status.idle":"2021-06-05T15:16:00.835669Z","shell.execute_reply.started":"2021-06-05T15:16:00.823766Z","shell.execute_reply":"2021-06-05T15:16:00.830858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(dataset_train)\nprint(dataset_train.__dict__)\nprint(dir(dataset_train))\n\n\n# Show annotation(s) for a DICOM image \ntest_fp = random.choice(image_fps_train)\nimage_annotations[test_fp]\n\n# prepare the validation dataset\ndataset_val = DetectorDataset(image_fps_val, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_val.prepare()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:16:04.552991Z","iopub.execute_input":"2021-06-05T15:16:04.553307Z","iopub.status.idle":"2021-06-05T15:16:05.70604Z","shell.execute_reply.started":"2021-06-05T15:16:04.553277Z","shell.execute_reply":"2021-06-05T15:16:05.705198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and display random samples and their bounding boxes\n# Suggestion: Run this a few times to see different examples. \n\nimage_id = random.choice(dataset_train.image_ids)\nimage_fp = dataset_train.image_reference(image_id)\n\n\nimage = dataset_train.load_image(image_id)\n##print(image)\nmask, class_ids = dataset_train.load_mask(image_id)\n##print(image_id)\nprint(mask.shape)\n####image.resize((1024, 1024,3))\n##print(mask)\nplt.figure(figsize=(10, 10))\nplt.subplot(1, 2, 1)\nplt.imshow(image[:, :, 0], cmap='gray')\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nmasked = np.zeros(image.shape[:2])\n#3print(masked)\nfor i in range(mask.shape[2]):\n    masked += image[:, :, 0] * mask[:, :, i]\nplt.imshow(masked, cmap='gray')\nplt.axis('off')\n\nprint(image_fp)\nprint(class_ids)","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:16:41.928541Z","iopub.execute_input":"2021-06-05T15:16:41.928891Z","iopub.status.idle":"2021-06-05T15:16:43.603572Z","shell.execute_reply.started":"2021-06-05T15:16:41.92886Z","shell.execute_reply":"2021-06-05T15:16:43.601691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----------------------------------Another way\n","metadata":{}},{"cell_type":"code","source":"train_image_path = os.path.join('../input/siim-covid19-detection/train')\ntest_image_path = os.path.join('../input/siim-covid19-detection/test')","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:17:00.668407Z","iopub.execute_input":"2021-06-05T15:17:00.66877Z","iopub.status.idle":"2021-06-05T15:17:00.673006Z","shell.execute_reply.started":"2021-06-05T15:17:00.668739Z","shell.execute_reply":"2021-06-05T15:17:00.671867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n    \n    def load_labels(self, labels_list):\n        for i, label in enumerate(labels_list):\n            self.add_class('covid', i + 1, label)\n            \n    def load_dataset(self, images_obj):\n        for image_obj in images_obj:\n            image_id = image_obj['image_id']\n            image_path = image_obj['image_path']\n            num_ids = image_obj['num_ids']\n            polygons = image_obj['polygons']\n            width = image_obj['width']\n            height = image_obj['height']\n            self.add_image(\"covid\", image_id=image_id, path=image_path,\n                           width=width, height=height, polygons=polygons,num_ids=num_ids)\n            \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def draw_shape(self, image, shape, dims, color):\n        \"\"\"Draws a shape from the given specs.\"\"\"\n        # Get the center x, y and the size s\n        x, y, s = dims\n        if shape == 'square':\n            cv2.rectangle(image, (x-s, y-s), (x+s, y+s), color, -1)\n        elif shape == \"circle\":\n            cv2.circle(image, (x, y), s, color, -1)\n        elif shape == \"triangle\":\n            points = np.array([[(x, y-s),\n                                (x-s/math.sin(math.radians(60)), y+s),\n                                (x+s/math.sin(math.radians(60)), y+s),\n                                ]], dtype=np.int32)\n            cv2.fillPoly(image, points, color)\n        return image\n\n    def load_mask(self, image_id):\n        \"\"\"Generate instance masks for an image.\n       Returns:\n        masks: A bool array of shape [height, width, instance count] with\n            one mask per instance.\n        class_ids: a 1D array of class IDs of the instance masks.\n        \"\"\"\n        info = self.image_info[image_id]\n        num_ids = info['num_ids']\n        mask = np.zeros([info[\"height\"], info[\"width\"], len(info[\"polygons\"])],\n                        dtype=np.uint8)\n\n        for i, p in enumerate(info[\"polygons\"]):\n            # Get indexes of pixels inside the polygon and set them to 1\n            rr, cc = skimage.draw.polygon(p['all_points_y'], p['all_points_x'])\n            mask[rr, cc, i] = 1\n\n        num_ids = np.array(num_ids, dtype=np.int32)\n        return mask, num_ids","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:17:05.756898Z","iopub.execute_input":"2021-06-05T15:17:05.757216Z","iopub.status.idle":"2021-06-05T15:17:05.770218Z","shell.execute_reply.started":"2021-06-05T15:17:05.757184Z","shell.execute_reply":"2021-06-05T15:17:05.769086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install xmltodict","metadata":{"execution":{"iopub.status.busy":"2021-06-05T14:29:50.644061Z","iopub.execute_input":"2021-06-05T14:29:50.644469Z","iopub.status.idle":"2021-06-05T14:29:57.998173Z","shell.execute_reply.started":"2021-06-05T14:29:50.644362Z","shell.execute_reply":"2021-06-05T14:29:57.997176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = [\"negetive\", \"Typical\", \"Indeterminate\",\"Atypical\"]","metadata":{"execution":{"iopub.status.busy":"2021-06-05T14:31:24.284876Z","iopub.execute_input":"2021-06-05T14:31:24.285217Z","iopub.status.idle":"2021-06-05T14:31:24.289495Z","shell.execute_reply.started":"2021-06-05T14:31:24.285182Z","shell.execute_reply":"2021-06-05T14:31:24.288651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_single_annotation(label_obj):\n    #print(label_obj)\n    name = label_obj['name']\n    # Get label\n    num_id = labels.index(name) + 1\n    bb_box = label_obj['bndbox']\n    # Extract the xmin xmax ymin and ymax of bounding box\n    xmin = int(bb_box['xmin'])\n    xmax = int(bb_box['xmax'])\n    ymin = int(bb_box['ymin'])\n    ymax = int(bb_box['ymax'])\n    # Convert it into polygon format. So we need 5 points for both x and y\n    all_points_x = [xmin, xmax, xmax, xmin, xmin]\n    all_points_y = [ymin, ymin, ymax, ymax, ymin]\n    return all_points_x, all_points_y, num_id","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:17:27.035104Z","iopub.execute_input":"2021-06-05T15:17:27.035463Z","iopub.status.idle":"2021-06-05T15:17:27.041014Z","shell.execute_reply.started":"2021-06-05T15:17:27.035407Z","shell.execute_reply":"2021-06-05T15:17:27.040068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xmltodict\nimport json\ntrain_images = []\ndef transform_annotations(utils.Dataset):\n    # Start the index from 100\n    curr_idx = 100\n    images_list = []\n    # List the files in the training or test path\n    for i in os.listdir(os.path.join(image_path)):\n        # Get the image path\n        img_path = os.path.join(image_path, i)\n        split_img_path = i.split('.')\n        # check if the file is a .jpg ext. We ignore .xml file as they will be parsed based on .jpg file name\n        if split_img_path[1] == 'dcm':\n            # Define dict key value pair required in coco dataset\n            polygons = []\n            num_ids = []\n            # Read the image file \n            file_data = cv2.imread(img_path)\n            # Get the heigh and width. OpenCV shape is in the format h, w, depth\n            height, width, _ = file_data.shape\n            # Open the xml file which has the same name of the image we have opened for this iteration\n            with open(os.path.join(image_path, split_img_path[0] + '.xml')) as fd:\n                # Load the xml -> convert xml to dict -> convert to json\n                bb_file = json.loads(json.dumps(xmltodict.parse(fd.read())))\n                # There are two case - bb_file['annotation']['object'] can exist as a single dict or as a list of dict.\n                # Thus, we need to do a check to see whether it is a list or not.\n                # If the value is a data type of list:\n                if isinstance(bb_file['annotation']['object'], list):\n                    # Loop through each dict in the list\n                    for obj in bb_file['annotation']['object']:\n                        # Parse each annotation individually\n                        all_points_x, all_points_y, num_id = parse_single_annotation(obj)\n                        # Append the points into polygon list\n                        polygons.append({\n                            'all_points_x': all_points_x,\n                            'all_points_y': all_points_y\n                        })\n                        # Append the id into the num_ids list\n                        num_ids.append(num_id)\n                # If the ['object'] key only contains a dict value\n                else:\n                    # We just need to parse a single annotation\n                    all_points_x, all_points_y, num_id = parse_single_annotation(bb_file['annotation']['object'])\n                    # Append it into polygon and num_ids list\n                    polygons.append({\n                        'all_points_x': all_points_x,\n                        'all_points_y': all_points_y\n                    })\n                    num_ids.append(num_id)\n            # For this image, we need to create a dict to represent it and all the corresponding annotations represented by polygons and num_ids key list\n            image_label = {\n                'image_path': img_path,\n                'image_id': curr_idx,\n                'polygons': polygons,\n                'num_ids': num_ids,\n                'height': height,\n                'width': width\n            }\n            curr_idx = curr_idx + 1\n            # Append it into the images_list\n            images_list.append(image_label)\n    return images_list","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:17:33.017132Z","iopub.execute_input":"2021-06-05T15:17:33.017484Z","iopub.status.idle":"2021-06-05T15:17:33.028431Z","shell.execute_reply.started":"2021-06-05T15:17:33.017453Z","shell.execute_reply":"2021-06-05T15:17:33.026807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = transform_annotations(image_annotations)\nprint(train_images[0:5])\ndataset_train = DetectorDataset()\ndataset_train.load_labels(labels)\ndataset_train.load_dataset(train_images)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T15:17:47.224387Z","iopub.execute_input":"2021-06-05T15:17:47.224737Z","iopub.status.idle":"2021-06-05T15:17:47.280695Z","shell.execute_reply.started":"2021-06-05T15:17:47.224707Z","shell.execute_reply":"2021-06-05T15:17:47.278993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! conda install -c conda-forge gdcm -y","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:09:48.343671Z","iopub.execute_input":"2021-06-07T19:09:48.344026Z","iopub.status.idle":"2021-06-07T19:10:51.908322Z","shell.execute_reply.started":"2021-06-07T19:09:48.343911Z","shell.execute_reply":"2021-06-07T19:10:51.907194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nIMG_FORMAT = \".dcm\"\nIMG_PATHS = []\nIMAGE_IDS = []\nIMAGE_NAMES = []\nSETS = []\nSERIES = []\nSTUDIES = []\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith(IMG_FORMAT):\n            img_path = os.path.join(dirname, filename)\n            Splitted = img_path.split('/')\n            # print(Splitted)\n            img_name = os.path.basename(img_path)\n            img_id = img_name.rstrip(IMG_FORMAT)\n            series_name = Splitted[-2]\n            study_name = Splitted[-3]\n            set_name = Splitted[-4]\n            IMG_PATHS.append(img_path)\n            IMAGE_NAMES.append(img_name)\n            IMAGE_IDS.append(img_id)\n            SETS.append(set_name)\n            SERIES.append(series_name)\n            STUDIES.append(study_name)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:11:12.715261Z","iopub.execute_input":"2021-06-07T19:11:12.715588Z","iopub.status.idle":"2021-06-07T19:11:40.482731Z","shell.execute_reply.started":"2021-06-07T19:11:12.715560Z","shell.execute_reply":"2021-06-07T19:11:40.481893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext = pd.DataFrame.from_dict({\"Image_Path\":IMG_PATHS,\n                             \"Image_Name\":IMAGE_NAMES,\n                             \"Image_ID\":IMAGE_IDS,\n                             \"Set_Name\": SETS,\n                             \"Series_Name\":SERIES,\n                             \"Study_Name\":STUDIES})\ndf_ext.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:21.682537Z","iopub.execute_input":"2021-06-07T19:19:21.682873Z","iopub.status.idle":"2021-06-07T19:19:21.720185Z","shell.execute_reply.started":"2021-06-07T19:19:21.682835Z","shell.execute_reply":"2021-06-07T19:19:21.719243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_lvl_pth = \"../input/siim-covid19-detection/train_image_level.csv\"\ndf_img = pd.read_csv(img_lvl_pth)\ndf_img.sort_values(by=['id'],inplace=True)\ndf_img.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:25.450082Z","iopub.execute_input":"2021-06-07T19:19:25.450401Z","iopub.status.idle":"2021-06-07T19:19:25.510896Z","shell.execute_reply.started":"2021-06-07T19:19:25.450367Z","shell.execute_reply":"2021-06-07T19:19:25.509955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_lvl_pth = \"../input/siim-covid19-detection/train_study_level.csv\"\ndf_std = pd.read_csv(std_lvl_pth)\ndf_std['id'] = df_std['id'].str.replace('_study',\"\")\ndf_std.rename({'id': 'StudyInstanceUID'},axis=1, inplace=True)\ndf_std.head(3)\n# df_std.sort_values(by=['StudyInstanceUID'],inplace=True)\ndf_std.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:29.737311Z","iopub.execute_input":"2021-06-07T19:19:29.737617Z","iopub.status.idle":"2021-06-07T19:19:29.772945Z","shell.execute_reply.started":"2021-06-07T19:19:29.737590Z","shell.execute_reply":"2021-06-07T19:19:29.772224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pth = \"../input/siim-covid19-detection/sample_submission.csv\"\ndf_sub = pd.read_csv(sub_pth)\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:33.479130Z","iopub.execute_input":"2021-06-07T19:19:33.479455Z","iopub.status.idle":"2021-06-07T19:19:33.503258Z","shell.execute_reply.started":"2021-06-07T19:19:33.479426Z","shell.execute_reply":"2021-06-07T19:19:33.502511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_img.merge(df_std, on='StudyInstanceUID')\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:36.858098Z","iopub.execute_input":"2021-06-07T19:19:36.858414Z","iopub.status.idle":"2021-06-07T19:19:36.881695Z","shell.execute_reply.started":"2021-06-07T19:19:36.858387Z","shell.execute_reply":"2021-06-07T19:19:36.880671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from copy import deepcopy\ndf_train = deepcopy(df)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:41.091329Z","iopub.execute_input":"2021-06-07T19:19:41.091649Z","iopub.status.idle":"2021-06-07T19:19:41.095845Z","shell.execute_reply.started":"2021-06-07T19:19:41.091618Z","shell.execute_reply":"2021-06-07T19:19:41.095059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"string = \"a29c5a68b07b\"\nstring.zfill(15)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:45.975923Z","iopub.execute_input":"2021-06-07T19:19:45.976280Z","iopub.status.idle":"2021-06-07T19:19:45.981766Z","shell.execute_reply.started":"2021-06-07T19:19:45.976249Z","shell.execute_reply":"2021-06-07T19:19:45.980710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:50.656428Z","iopub.execute_input":"2021-06-07T19:19:50.656743Z","iopub.status.idle":"2021-06-07T19:19:50.663547Z","shell.execute_reply.started":"2021-06-07T19:19:50.656715Z","shell.execute_reply":"2021-06-07T19:19:50.662451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext[\"id\"] = df_ext[\"Image_Name\"].str.zfill(19)\ndf_ext.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:54.257310Z","iopub.execute_input":"2021-06-07T19:19:54.257613Z","iopub.status.idle":"2021-06-07T19:19:54.268654Z","shell.execute_reply.started":"2021-06-07T19:19:54.257589Z","shell.execute_reply":"2021-06-07T19:19:54.267670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext[\"Set_Name\"].value_counts()\ndf_test = df_ext[df_ext[\"Set_Name\"]==\"test\"]\ndf_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:19:57.450308Z","iopub.execute_input":"2021-06-07T19:19:57.450621Z","iopub.status.idle":"2021-06-07T19:19:57.478601Z","shell.execute_reply.started":"2021-06-07T19:19:57.450593Z","shell.execute_reply":"2021-06-07T19:19:57.477641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test = deepcopy(df_train)\nCOLS = list(df_train_test.columns)\ndef subtract_lists(x,y):\n    \"\"\"Subtract Two Lists (List Difference)\"\"\"\n    return [item for item in x if item not in y]\ndef merge_list_to_dict(test_keys,test_values):\n    \"\"\"Using dictionary comprehension to merge two lists to dictionary\"\"\"\n    merged_dict = {test_keys[i]: test_values[i] for i in range(len(test_keys))}\n    return merged_dict\n# NAN_COLS = subtract_lists(COLS,[\"id\"])\nTO_ATTACH = merge_list_to_dict(COLS,[np.nan]*len(COLS))\nfor index, row in df_test.iterrows():\n    TO_ATTACH[\"id\"] = row[\"id\"]\n    df_train_test = df_train_test.append(TO_ATTACH, ignore_index = True)\ndf_train_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:20:01.602310Z","iopub.execute_input":"2021-06-07T19:20:01.602617Z","iopub.status.idle":"2021-06-07T19:20:06.571372Z","shell.execute_reply.started":"2021-06-07T19:20:01.602590Z","shell.execute_reply":"2021-06-07T19:20:06.570440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext[\"id\"] = df_ext[\"id\"].str.zfill(19)\ndf_ext.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:20:16.257165Z","iopub.execute_input":"2021-06-07T19:20:16.257476Z","iopub.status.idle":"2021-06-07T19:20:16.274205Z","shell.execute_reply.started":"2021-06-07T19:20:16.257449Z","shell.execute_reply":"2021-06-07T19:20:16.273215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_ext.merge(df_train_test, on='id')\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:20:22.356004Z","iopub.execute_input":"2021-06-07T19:20:22.356326Z","iopub.status.idle":"2021-06-07T19:20:22.385526Z","shell.execute_reply.started":"2021-06-07T19:20:22.356298Z","shell.execute_reply":"2021-06-07T19:20:22.384528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nwarnings.filterwarnings('ignore')\nimport shutil \nimport tensorflow as tf\n%matplotlib inline\n\n\nimport matplotlib.pylab as pylab\nimport seaborn as sns\nimport pprint\nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport wandb\n\nimport PIL\nfrom PIL import Image\nfrom colorama import Fore, Back, Style\nviz_counter=0\n\ndef create_dir(dir, v=1):\n    \"\"\"\n    Creates a directory without throwing an error if directory already exists.\n    dir : The directory to be created.\n    v : Verbosity\n    \"\"\"\n    if not os.path.exists(dir):\n        os.makedirs(dir)\n        if v:\n            print(\"Created Directory : \", dir)\n        return 1\n    else:\n        if v:\n            print(\"Directory already existed : \", dir)\n        return 0\n\nvoi_lut=True\nfix_monochrome=True\n\ndef dicom_dataset_to_dict(filename):\n    \"\"\"Credit: https://github.com/pydicom/pydicom/issues/319\n               https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    \"\"\"\n    \n    dicom_header = dicom.dcmread(filename) \n    \n    #====== DICOM FILE DATA ======\n    dicom_dict = {}\n    repr(dicom_header)\n    for dicom_value in dicom_header.values():\n        if dicom_value.tag == (0x7fe0, 0x0010):\n            #discard pixel data\n            continue\n        if type(dicom_value.value) == dicom.dataset.Dataset:\n            dicom_dict[dicom_value.name] = dicom_dataset_to_dict(dicom_value.value)\n        else:\n            v = _convert_value(dicom_value.value)\n            dicom_dict[dicom_value.name] = v\n      \n    del dicom_dict['Pixel Representation']\n    \n    #====== DICOM IMAGE DATA ======\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom_header.pixel_array, dicom_header)\n    else:\n        data = dicom_header.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom_header.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    modified_image_data = (data * 255).astype(np.uint8)\n    \n    return dicom_dict, modified_image_data\n\ndef _sanitise_unicode(s):\n    return s.replace(u\"\\u0000\", \"\").strip()\n\ndef _convert_value(v):\n    t = type(v)\n    if t in (list, int, float):\n        cv = v\n    elif t == str:\n        cv = _sanitise_unicode(v)\n    elif t == bytes:\n        s = v.decode('ascii', 'replace')\n        cv = _sanitise_unicode(s)\n    elif t == dicom.valuerep.DSfloat:\n        cv = float(v)\n    elif t == dicom.valuerep.IS:\n        cv = int(v)\n    else:\n        cv = repr(v)\n    return cv\n\n\nimport os, fnmatch\ndef find(pattern, path):\n    \"\"\"Utility to find files wrt a regex search\"\"\"\n    result = []\n    for root, dirs, files in os.walk(path):\n        for name in files:\n            if fnmatch.fnmatch(name, pattern):\n                result.append(os.path.join(root, name))\n    return result\ndef props(arr):\n    print(\"Shape :\",arr.shape,\"Maximum :\",arr.max(),\"Minimum :\",arr.min(),\"Data Type :\",arr.dtype)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:20:31.214756Z","iopub.execute_input":"2021-06-07T19:20:31.215104Z","iopub.status.idle":"2021-06-07T19:20:36.840258Z","shell.execute_reply.started":"2021-06-07T19:20:31.215071Z","shell.execute_reply":"2021-06-07T19:20:36.839298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nShapes that you wish to resize to\n\"\"\"\n\nShape_X = 1024\nShape_Y = 1024\nimage_id = []\ndim0 = []\ndim1 = []\nsplits = []\nimg_paths = []\n\nfor split in ['test', 'train']:\n    # save_dir = f'/kaggle/tmp/{split}/'\n    save_dir = f'/kaggle/working/resized_data/{split}/'\n    print(split)\n    os.makedirs(save_dir, exist_ok=True)\n    \n    for dirname, _, filenames in tqdm(os.walk(f'/kaggle/input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            fpath = os.path.join(dirname, file)\n            dicom_dict, modified_image_data = dicom_dataset_to_dict(fpath)\n            res = cv2.resize(modified_image_data,(Shape_Y,Shape_X)) # cv2 has this opposite\n            save_path = os.path.join(save_dir, file.replace('dcm', 'png'))\n            cv2.imwrite(save_path,res)\n            img_id = file.replace('.dcm', '')\n            image_id.append(img_id)\n            dim0.append(modified_image_data.shape[0])\n            dim1.append(modified_image_data.shape[1])\n            img_paths.append(fpath)\n            splits.append(split)\n\"\"\"\n2475/?\n12386/?\n07:34 | 5.38it/s\n36:51 | 8.13it/s\n\"\"\"\nprint(\"Generation Complete!\")\n","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:20:52.560741Z","iopub.execute_input":"2021-06-07T19:20:52.561183Z","iopub.status.idle":"2021-06-07T19:59:54.403667Z","shell.execute_reply.started":"2021-06-07T19:20:52.561141Z","shell.execute_reply":"2021-06-07T19:59:54.402708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df = pd.DataFrame.from_dict({'Image_Path': img_paths, 'dim0': dim0, 'dim1': dim1})\nnew_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T19:59:54.405491Z","iopub.execute_input":"2021-06-07T19:59:54.406155Z","iopub.status.idle":"2021-06-07T19:59:54.432389Z","shell.execute_reply.started":"2021-06-07T19:59:54.406114Z","shell.execute_reply":"2021-06-07T19:59:54.431443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df = df.merge(new_df,on=\"Image_Path\")\nfinal_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:11.444466Z","iopub.execute_input":"2021-06-07T20:13:11.444789Z","iopub.status.idle":"2021-06-07T20:13:11.479748Z","shell.execute_reply.started":"2021-06-07T20:13:11.444762Z","shell.execute_reply":"2021-06-07T20:13:11.478963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ast import literal_eval","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:25.828209Z","iopub.execute_input":"2021-06-07T20:13:25.828523Z","iopub.status.idle":"2021-06-07T20:13:25.832872Z","shell.execute_reply.started":"2021-06-07T20:13:25.828496Z","shell.execute_reply":"2021-06-07T20:13:25.831679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = len(final_df)\n# Already defined above : Shape_Y,Shape_X = 512,512\nNEW_BOXES = []\nfor i in range(n):\n    if type(final_df['boxes'][i])==str:\n        boxes = literal_eval(final_df['boxes'][i])\n        BIG_BOX = []\n        for box in boxes:\n            xbase,ybase = (box['x']*(Shape_Y/final_df['dim1'][i]), box['y']*(Shape_X/final_df['dim0'][i]))\n            new_width,new_height = box['width']*(Shape_Y/final_df['dim1'][i]), box['height']*(Shape_X/final_df['dim0'][i])\n            CURR_BOX = {\"x\": xbase,\n                        \"y\" : ybase,\n                        \"width\" : new_width,\n                        \"height\" : new_height}\n            BIG_BOX.append(CURR_BOX)\n    else:\n        BIG_BOX = \"\"\n    NEW_BOXES.append(str(BIG_BOX))","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:32.153082Z","iopub.execute_input":"2021-06-07T20:13:32.153453Z","iopub.status.idle":"2021-06-07T20:13:32.175814Z","shell.execute_reply.started":"2021-06-07T20:13:32.153420Z","shell.execute_reply":"2021-06-07T20:13:32.174737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df['corrected_boxes'] = NEW_BOXES","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:38.517394Z","iopub.execute_input":"2021-06-07T20:13:38.517745Z","iopub.status.idle":"2021-06-07T20:13:38.523631Z","shell.execute_reply.started":"2021-06-07T20:13:38.517702Z","shell.execute_reply":"2021-06-07T20:13:38.522791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correction Factors\nfinal_df['cfy'] = Shape_Y/final_df['dim1']\nfinal_df['cfx'] = Shape_X/final_df['dim0']","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:41.337326Z","iopub.execute_input":"2021-06-07T20:13:41.337637Z","iopub.status.idle":"2021-06-07T20:13:41.431804Z","shell.execute_reply.started":"2021-06-07T20:13:41.337610Z","shell.execute_reply":"2021-06-07T20:13:41.431067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install openpyxl","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:45.297550Z","iopub.execute_input":"2021-06-07T20:13:45.297858Z","iopub.status.idle":"2021-06-07T20:13:53.499059Z","shell.execute_reply.started":"2021-06-07T20:13:45.297830Z","shell.execute_reply":"2021-06-07T20:13:53.498121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('Extracted_Study_Series_Img.csv',index=False)\nfinal_df.to_excel('Extracted_Study_Series_Img.xlsx',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:13:53.500731Z","iopub.execute_input":"2021-06-07T20:13:53.501064Z","iopub.status.idle":"2021-06-07T20:13:54.449129Z","shell.execute_reply.started":"2021-06-07T20:13:53.501036Z","shell.execute_reply":"2021-06-07T20:13:54.448241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nimport sys\nimport random\nimport math\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport json\nimport pydicom\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport pandas as pd \nimport glob ","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:14:37.408424Z","iopub.execute_input":"2021-06-07T20:14:37.408753Z","iopub.status.idle":"2021-06-07T20:14:38.291263Z","shell.execute_reply.started":"2021-06-07T20:14:37.408724Z","shell.execute_reply":"2021-06-07T20:14:38.290426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = '/kaggle/working'\n\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:14:58.852448Z","iopub.execute_input":"2021-06-07T20:14:58.852772Z","iopub.status.idle":"2021-06-07T20:14:58.856914Z","shell.execute_reply.started":"2021-06-07T20:14:58.852743Z","shell.execute_reply":"2021-06-07T20:14:58.855545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\nos.chdir('Mask_RCNN')\n#!python setup.py -q install","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:20:31.729882Z","iopub.execute_input":"2021-06-07T20:20:31.730334Z","iopub.status.idle":"2021-06-07T20:20:40.247404Z","shell.execute_reply.started":"2021-06-07T20:20:31.730296Z","shell.execute_reply":"2021-06-07T20:20:40.246430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Mask RCNN\nsys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:20:46.539696Z","iopub.execute_input":"2021-06-07T20:20:46.540053Z","iopub.status.idle":"2021-06-07T20:20:46.660642Z","shell.execute_reply.started":"2021-06-07T20:20:46.540018Z","shell.execute_reply":"2021-06-07T20:20:46.659835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.png')\n    return list(set(dicom_fps))\n\ndef parse_dataset(dicom_dir, anns): \n    image_fps = get_dicom_fps(dicom_dir)\n    ##print(image_fps)\n    image_annotations = {}\n    for index, row in anns.iterrows(): \n        fp = os.path.join(dicom_dir, row['id']+'.png')\n        ##print(fp)\n        ##print(row)\n        image_annotations[fp]=(row)\n    return image_fps, image_annotations ","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:20:50.119515Z","iopub.execute_input":"2021-06-07T20:20:50.119831Z","iopub.status.idle":"2021-06-07T20:20:50.126467Z","shell.execute_reply.started":"2021-06-07T20:20:50.119802Z","shell.execute_reply":"2021-06-07T20:20:50.125420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass DetectorConfig(Config):\n    \"\"\"Configuration for training pneumonia detection on the RSNA pneumonia dataset.\n    Overrides values in the base Config class.\n    \"\"\"\n    \n    # Give the configuration a recognizable name  \n    NAME = 'covid'\n    \n    # Train on 1 GPU and 8 images per GPU. We can put multiple images on each\n    # GPU because the images are small. Batch size is 8 (GPUs * images/GPU).\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 8 \n    \n    BACKBONE = 'resnet50'\n    \n    NUM_CLASSES = 1+4  # background + 1 pneumonia classes\n    \n    IMAGE_MIN_DIM = 256\n    IMAGE_MAX_DIM = 256\n    RPN_ANCHOR_SCALES = (32, 64, 128, 256)\n    TRAIN_ROIS_PER_IMAGE = 32\n    MAX_GT_INSTANCES = 3\n    DETECTION_MAX_INSTANCES = 3\n    DETECTION_MIN_CONFIDENCE = 0.9\n    DETECTION_NMS_THRESHOLD = 0.1\n\n    STEPS_PER_EPOCH = 100\n    \nconfig = DetectorConfig()\nconfig.display()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:21:02.637531Z","iopub.execute_input":"2021-06-07T20:21:02.637845Z","iopub.status.idle":"2021-06-07T20:21:02.661932Z","shell.execute_reply.started":"2021-06-07T20:21:02.637815Z","shell.execute_reply":"2021-06-07T20:21:02.660723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, image_fps, image_annotations, orig_height, orig_width):\n        super().__init__(self)\n        \n        # Add classes\n        self.add_class('covid', 1, 'negetive')\n        self.add_class('covid', 2, 'typical')\n        self.add_class('covid', 3, 'intermediate')\n        self.add_class('covid', 4, 'atypical')\n   \n        # add images \n        for i, fp in enumerate(image_fps):\n            annotations = image_annotations[fp]\n            self.add_image('covid', image_id=i, path=fp, \n                           annotations=annotations, orig_height=orig_height, orig_width=orig_width)\n            \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        fp = info['path']\n        ds = pydicom.read_file(fp)\n        image = ds.pixel_array\n        # If grayscale. Convert to RGB for consistency.\n        if len(image.shape) != 3 or image.shape[2] != 3:\n            image = np.stack((image,) * 3, -1)\n        return image\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        annotations = info['annotations']\n        count = len(annotations)\n        if count == 0:\n            mask = np.zeros((info['orig_height'], info['orig_width'], 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            mask = np.zeros((info['orig_height'], info['orig_width'], count), dtype=np.uint8)\n            class_ids = np.zeros((count,), dtype=np.int32)\n            for i, a in enumerate(annotations):\n                    x = int(a['x'])\n                    y = int(a['y'])\n                    w = int(a['width'])\n                    h = int(a['height'])\n                    mask_instance = mask[:, :, i].copy()\n                    cv2.rectangle(mask_instance, (x, y), (x+w, y+h), 255, -1)\n                    mask[:, :, i] = mask_instance\n                    class_ids[i] = i\n        return mask.astype(np.bool), class_ids.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:21:59.734752Z","iopub.execute_input":"2021-06-07T20:21:59.735079Z","iopub.status.idle":"2021-06-07T20:21:59.749083Z","shell.execute_reply.started":"2021-06-07T20:21:59.735047Z","shell.execute_reply":"2021-06-07T20:21:59.748045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_fps, image_annotations = parse_dataset(train_dicom_dir, anns=anns)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DTA_DIR='/kaggle/input/siim-covid19-updated-train-labels'\nanns = pd.read_csv(os.path.join(DTA_DIR, 'updated_train_labels.csv'))\nanns.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:23:06.089282Z","iopub.execute_input":"2021-06-07T20:23:06.089605Z","iopub.status.idle":"2021-06-07T20:23:06.189839Z","shell.execute_reply.started":"2021-06-07T20:23:06.089577Z","shell.execute_reply":"2021-06-07T20:23:06.188881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dicom_dir = os.path.join(DATA_DIR, 'resized_data/train')\ntest_dicom_dir = os.path.join(DATA_DIR, 'resized_data/test')\nimage_fps, image_annotations = parse_dataset(train_dicom_dir, anns=anns)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:25:12.221476Z","iopub.execute_input":"2021-06-07T20:25:12.221836Z","iopub.status.idle":"2021-06-07T20:25:12.955380Z","shell.execute_reply.started":"2021-06-07T20:25:12.221804Z","shell.execute_reply":"2021-06-07T20:25:12.954513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_annotations)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:25:44.219565Z","iopub.execute_input":"2021-06-07T20:25:44.219908Z","iopub.status.idle":"2021-06-07T20:25:49.105013Z","shell.execute_reply.started":"2021-06-07T20:25:44.219876Z","shell.execute_reply":"2021-06-07T20:25:49.103717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nimage = cv2.imread(image_fps[1])\nplt.imshow(image)\nplt.show()\ndata = np.asarray( image, dtype='uint8' )\ndata","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:29:03.086768Z","iopub.execute_input":"2021-06-07T20:29:03.087105Z","iopub.status.idle":"2021-06-07T20:29:03.393412Z","shell.execute_reply.started":"2021-06-07T20:29:03.087075Z","shell.execute_reply":"2021-06-07T20:29:03.392451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Original DICOM image size: 1024 x 1024\nORIG_SIZE = 1024","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:29:38.496724Z","iopub.execute_input":"2021-06-07T20:29:38.497121Z","iopub.status.idle":"2021-06-07T20:29:38.502447Z","shell.execute_reply.started":"2021-06-07T20:29:38.497088Z","shell.execute_reply":"2021-06-07T20:29:38.501519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_fps_list = list(image_fps[:1000]) \n#####################################################################\n\n# split dataset into training vs. validation dataset \n# split ratio is set to 0.9 vs. 0.1 (train vs. validation, respectively)\nsorted(image_fps_list)\nrandom.seed(42)\nrandom.shuffle(image_fps_list)\n\nvalidation_split = 0.1\nsplit_index = int((1 - validation_split) * len(image_fps_list))\n\nimage_fps_train = image_fps_list[:split_index]\nimage_fps_val = image_fps_list[split_index:]\n\nprint(len(image_fps_train), len(image_fps_val))","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:29:54.876766Z","iopub.execute_input":"2021-06-07T20:29:54.877108Z","iopub.status.idle":"2021-06-07T20:29:54.884370Z","shell.execute_reply.started":"2021-06-07T20:29:54.877076Z","shell.execute_reply":"2021-06-07T20:29:54.883310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train = DetectorDataset(image_fps_train, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:30:19.575715Z","iopub.execute_input":"2021-06-07T20:30:19.576065Z","iopub.status.idle":"2021-06-07T20:30:19.583115Z","shell.execute_reply.started":"2021-06-07T20:30:19.576034Z","shell.execute_reply":"2021-06-07T20:30:19.582143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_fp = random.choice(image_fps_train)\nimage_annotations[test_fp]","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:30:53.907474Z","iopub.execute_input":"2021-06-07T20:30:53.907864Z","iopub.status.idle":"2021-06-07T20:30:53.916345Z","shell.execute_reply.started":"2021-06-07T20:30:53.907833Z","shell.execute_reply":"2021-06-07T20:30:53.915442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_val = DetectorDataset(image_fps_val, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_val.prepare()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:31:33.116875Z","iopub.execute_input":"2021-06-07T20:31:33.117279Z","iopub.status.idle":"2021-06-07T20:31:33.124322Z","shell.execute_reply.started":"2021-06-07T20:31:33.117249Z","shell.execute_reply":"2021-06-07T20:31:33.123137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, image_fps, image_annotations, orig_height, orig_width):\n        super().__init__(self)\n        \n        # Add classes\n        self.add_class('covid', 1, 'negetive')\n        self.add_class('covid', 2, 'typical')\n        self.add_class('covid', 3, 'intermediate')\n        self.add_class('covid', 4, 'atypical')\n   \n        # add images \n        for i, fp in enumerate(image_fps):\n            annotations = image_annotations[fp]\n            self.add_image('covid', image_id=i, path=fp, \n                           annotations=annotations, orig_height=orig_height, orig_width=orig_width)\n            \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        fp = info['path']\n        image = cv2.imread(fp)\n        plt.imshow(image)\n        plt.show()\n        data = np.asarray( image, dtype='uint8' )\n        # If grayscale. Convert to RGB for consistency.\n        if len(data.shape) != 3 or data.shape[2] != 3:\n            data = np.stack((data,) * 3, -1)\n        return data\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        annotations = info['annotations']\n        W=info['orig_width']\n        H=info['orig_height']\n        count = len(annotations)\n        if count == 0:\n            mask = np.zeros((info['orig_height'], info['orig_width'], 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            mask = np.zeros((info['orig_height'], info['orig_width'], count), dtype=np.uint8)\n            class_ids = np.zeros((count,), dtype=np.int32)\n            for i, a in enumerate(annotations):\n                    ##x = (a[2])\n                    ##y = (a[3])\n                   ## print(x)\n                    ##print(y)\n                    ##w = a[4]\n                    ##h = a[5]\n                    mask_instance = mask[:, :, i].copy()\n                    \n                    x = int(float(a[2]))\n                    y = int(float(a[3]))\n                    w = int(float(a[4]))\n                    h = int(float(a[5]))\n\n                    ##cv2.rectangle(frame, (startX, startY), (endX, endY), (155, 255, 0), 2)\"\"\"\n                    cv2.rectangle(mask_instance, (x, y), (w, h), (155, 255, 0), 2)\n                    mask[:, :, i] = mask_instance\n                    class_ids[i] = i\n        return mask.astype(np.bool), class_ids.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T21:24:13.235049Z","iopub.execute_input":"2021-06-07T21:24:13.235393Z","iopub.status.idle":"2021-06-07T21:24:13.253608Z","shell.execute_reply.started":"2021-06-07T21:24:13.235364Z","shell.execute_reply":"2021-06-07T21:24:13.252475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_train = DetectorDataset(image_fps_train, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2021-06-07T21:24:18.621725Z","iopub.execute_input":"2021-06-07T21:24:18.622068Z","iopub.status.idle":"2021-06-07T21:24:18.628371Z","shell.execute_reply.started":"2021-06-07T21:24:18.622036Z","shell.execute_reply":"2021-06-07T21:24:18.627400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and display random samples and their bounding boxes\n# Suggestion: Run this a few times to see different examples. \n\nimage_id = random.choice(dataset_train.image_ids)\nimage_fp = dataset_train.image_reference(image_id)\nimage = dataset_train.load_image(image_id)\nmask, class_ids = dataset_train.load_mask(image_id)\n\nprint(image.shape)\n\nplt.figure(figsize=(10, 10))\nplt.subplot(1, 2, 1)\nplt.imshow(image[:, :, 0], cmap='gray')\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nmasked = np.zeros(image.shape[:2])\nfor i in range(mask.shape[2]):\n    masked += image[:, :, 0] * mask[:, :, i]\nplt.imshow(masked, cmap='gray')\nplt.axis('off')\n\nprint(image_fp)\nprint(class_ids)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T21:24:22.174004Z","iopub.execute_input":"2021-06-07T21:24:22.174334Z","iopub.status.idle":"2021-06-07T21:24:22.439204Z","shell.execute_reply.started":"2021-06-07T21:24:22.174304Z","shell.execute_reply":"2021-06-07T21:24:22.437560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)","metadata":{"execution":{"iopub.status.busy":"2021-06-07T20:33:24.137916Z","iopub.execute_input":"2021-06-07T20:33:24.138302Z","iopub.status.idle":"2021-06-07T20:33:24.260940Z","shell.execute_reply.started":"2021-06-07T20:33:24.138272Z","shell.execute_reply":"2021-06-07T20:33:24.259198Z"},"trusted":true},"execution_count":null,"outputs":[]}]}