{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Yet another EDA for train CSV files...\n\n## Preparations","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport glob\nimport pydicom","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-08T10:47:18.742495Z","iopub.execute_input":"2021-06-08T10:47:18.742935Z","iopub.status.idle":"2021-06-08T10:47:19.006969Z","shell.execute_reply.started":"2021-06-08T10:47:18.742854Z","shell.execute_reply":"2021-06-08T10:47:19.005663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"siim_covid19_dir = os.path.join(\n    '..', 'input', 'siim-covid19-detection')","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.011019Z","iopub.execute_input":"2021-06-08T10:47:19.011340Z","iopub.status.idle":"2021-06-08T10:47:19.016087Z","shell.execute_reply.started":"2021-06-08T10:47:19.011309Z","shell.execute_reply":"2021-06-08T10:47:19.014826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_csv(file_name):\n    file_path = os.path.join(siim_covid19_dir, file_name)\n    df = pd.read_csv(file_path)\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.017985Z","iopub.execute_input":"2021-06-08T10:47:19.018302Z","iopub.status.idle":"2021-06-08T10:47:19.029908Z","shell.execute_reply.started":"2021-06-08T10:47:19.018269Z","shell.execute_reply":"2021-06-08T10:47:19.028996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----\n## train_study_level.csv\n\n* 6054 rows x 5 columns\n* Columns: 'id' and 4 labels","metadata":{}},{"cell_type":"code","source":"study_level_df = read_csv('train_study_level.csv')\n\nstudy_level_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.031443Z","iopub.execute_input":"2021-06-08T10:47:19.031966Z","iopub.status.idle":"2021-06-08T10:47:19.089832Z","shell.execute_reply.started":"2021-06-08T10:47:19.031934Z","shell.execute_reply":"2021-06-08T10:47:19.088846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many rows?\nstudy_level_num_rows = len(study_level_df)\n\nstudy_level_num_rows","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.091215Z","iopub.execute_input":"2021-06-08T10:47:19.091489Z","iopub.status.idle":"2021-06-08T10:47:19.097138Z","shell.execute_reply.started":"2021-06-08T10:47:19.091462Z","shell.execute_reply":"2021-06-08T10:47:19.096442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Study Level: id\n\n* Unique and no duplicates.","metadata":{}},{"cell_type":"code","source":"# Are unique? Any duplicates?\nstudy_level_num_unique_ids = len(pd.unique(study_level_df['id']))\n\nif study_level_num_unique_ids == study_level_num_rows:\n    print(\"Unique and no duplicates\")\nelse:\n    print(\"Some duplicates\")","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.098324Z","iopub.execute_input":"2021-06-08T10:47:19.098801Z","iopub.status.idle":"2021-06-08T10:47:19.114805Z","shell.execute_reply.started":"2021-06-08T10:47:19.098767Z","shell.execute_reply":"2021-06-08T10:47:19.113733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Study Level: Labels\n\n* Only one of four label values is 1 in training data.\n* For test data, maybe multiple 1's ([Overview -- Evaluation](https://www.kaggle.com/c/siim-covid19-detection/overview/evaluation)).","metadata":{}},{"cell_type":"code","source":"# What are the unique value combinations? How many of them?\nstudy_level_label_colums = [\n    'Negative for Pneumonia', 'Typical Appearance',\n    'Indeterminate Appearance', 'Atypical Appearance' ]\nstudy_level_labels_df = study_level_df[study_level_label_colums]\nstudy_level_label_values = study_level_labels_df.values\n\nstudy_level_unique_label_combinations, \\\nstudy_level_unique_label_counts = \\\n    np.unique(\n        study_level_label_values, return_counts=True, axis=0)\n\nprint(\"Unique Combinations:\\n\", study_level_unique_label_combinations)\nprint(\"Unique Counts:\\n\", study_level_unique_label_counts)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.115972Z","iopub.execute_input":"2021-06-08T10:47:19.116545Z","iopub.status.idle":"2021-06-08T10:47:19.140430Z","shell.execute_reply.started":"2021-06-08T10:47:19.116505Z","shell.execute_reply":"2021-06-08T10:47:19.139167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----\n## train_image_level.csv\n\n* 6334 rows x 4 columns\n* Columns: 'id', 'boxes', 'label', and 'StudyInstanceUID'","metadata":{}},{"cell_type":"code","source":"image_level_df = read_csv('train_image_level.csv')\n\nimage_level_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.143449Z","iopub.execute_input":"2021-06-08T10:47:19.143788Z","iopub.status.idle":"2021-06-08T10:47:19.215267Z","shell.execute_reply.started":"2021-06-08T10:47:19.143758Z","shell.execute_reply":"2021-06-08T10:47:19.214072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many rows?\nimage_level_num_rows = len(image_level_df)\n\nimage_level_num_rows","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.218133Z","iopub.execute_input":"2021-06-08T10:47:19.218440Z","iopub.status.idle":"2021-06-08T10:47:19.225144Z","shell.execute_reply.started":"2021-06-08T10:47:19.218409Z","shell.execute_reply":"2021-06-08T10:47:19.223972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Level: id\n\n* Unique and no dupulicates","metadata":{}},{"cell_type":"code","source":"# Are unique? Any duplicates?\nimage_level_num_unique_ids = len(pd.unique(image_level_df['id']))\n\nif image_level_num_unique_ids == image_level_num_rows:\n    print(\"Unique and no duplicates\")\nelse:\n    print(\"Some duplicates\")","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.226661Z","iopub.execute_input":"2021-06-08T10:47:19.227018Z","iopub.status.idle":"2021-06-08T10:47:19.238518Z","shell.execute_reply.started":"2021-06-08T10:47:19.226986Z","shell.execute_reply":"2021-06-08T10:47:19.237794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Level: Boxes\n\n* A list of dictionaries. Each dictionary holds bbox information.\n* NaN for no bbox, empty list is easier to handle...","metadata":{}},{"cell_type":"code","source":"# How does it look like?\nfor i in range(5):\n    boxes = image_level_df.loc[i, 'boxes']\n    print(\"{0}: {1}\".format(i, boxes))","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.240124Z","iopub.execute_input":"2021-06-08T10:47:19.240453Z","iopub.status.idle":"2021-06-08T10:47:19.251705Z","shell.execute_reply.started":"2021-06-08T10:47:19.240412Z","shell.execute_reply":"2021-06-08T10:47:19.250775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Level: Label\n\n* Format: \"prediction, confidence, left, top, right, bottom\", ...\n* Number of fields is multiple of 6, maximum is 48.\n* Maximum number of bboxes for a image is 8 (= 48 / 6).","metadata":{}},{"cell_type":"code","source":"# How does it look like?\nfor i in range(5):\n    label = image_level_df.loc[ i, 'label']\n    print(\"{0}: {1}\".format(i, label))","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.252918Z","iopub.execute_input":"2021-06-08T10:47:19.253304Z","iopub.status.idle":"2021-06-08T10:47:19.262227Z","shell.execute_reply.started":"2021-06-08T10:47:19.253223Z","shell.execute_reply":"2021-06-08T10:47:19.261466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many rows have how many fields?\n# 3013 rows have 6 fileds, ..., 1 row has 48 fields.\nimage_level_label_field_counts = \\\n    image_level_df['label'] \\\n        .apply(lambda label: len(label.split())) \\\n        .value_counts() \\\n        .sort_index()\n\nprint(image_level_label_field_counts)\nassert sum(image_level_label_field_counts) == image_level_num_rows","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.263494Z","iopub.execute_input":"2021-06-08T10:47:19.264045Z","iopub.status.idle":"2021-06-08T10:47:19.286214Z","shell.execute_reply.started":"2021-06-08T10:47:19.264006Z","shell.execute_reply":"2021-06-08T10:47:19.285459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_image_level_label_df(label):\n    '''Make a DataFrame from a label string.'''\n    fields_list = label.split()\n    num_fields = len(fields_list)\n    # https://note.nkmk.me/python-list-ndarray-1d-to-2d/\n    fields_2d_list = [\n        fields_list[ i:i+6 ] for i in range(0, num_fields, 6)]\n    columns = [\n        'prediction', 'confidence',\n        'left', 'top', 'right', 'bottom']\n    label_df = pd.DataFrame(fields_2d_list, columns=columns)\n    label_df = label_df.astype({\n        'confidence': np.float32,\n        'left': np.float32, 'top': np.float32,\n        'right': np.float32, 'bottom': np.float32 })\n    return label_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.287336Z","iopub.execute_input":"2021-06-08T10:47:19.287792Z","iopub.status.idle":"2021-06-08T10:47:19.294590Z","shell.execute_reply.started":"2021-06-08T10:47:19.287751Z","shell.execute_reply":"2021-06-08T10:47:19.293717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Image Level: Label: Prediction\n\n* The prediction for each bbox is either 'opacity' or 'none'.\n* Number of 'opacity' is 7853, and 'none' is 2040.\n* For each image, prediction is either:\n    * 'none', or\n    * one or more 'opacity'.","metadata":{}},{"cell_type":"code","source":"# For each labels, what predictions and how many?\nimage_level_label_pred_count_dict_list = []\nfor idx, (image_id, label) in image_level_df[['id', 'label']].iterrows():\n    label_df = make_image_level_label_df(label)\n    pred_count_dict = label_df['prediction'].value_counts().to_dict()\n    image_level_label_pred_count_dict_list.append(pred_count_dict)\n    \nimage_level_label_pred_df = pd.DataFrame(\n    image_level_label_pred_count_dict_list)\nimage_level_label_pred_df = image_level_label_pred_df.fillna(0)\n\nimage_level_label_pred_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:19.295850Z","iopub.execute_input":"2021-06-08T10:47:19.296327Z","iopub.status.idle":"2021-06-08T10:47:39.678420Z","shell.execute_reply.started":"2021-06-08T10:47:19.296284Z","shell.execute_reply":"2021-06-08T10:47:39.677480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many for each predictions?\nimage_level_label_pred_df.sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:39.679756Z","iopub.execute_input":"2021-06-08T10:47:39.680306Z","iopub.status.idle":"2021-06-08T10:47:39.688400Z","shell.execute_reply.started":"2021-06-08T10:47:39.680262Z","shell.execute_reply":"2021-06-08T10:47:39.687647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many for each prediction combinations?\nimage_level_label_pred_df.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:39.689705Z","iopub.execute_input":"2021-06-08T10:47:39.690316Z","iopub.status.idle":"2021-06-08T10:47:39.709879Z","shell.execute_reply.started":"2021-06-08T10:47:39.690255Z","shell.execute_reply":"2021-06-08T10:47:39.708969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Image Level: StudyInstanceUID\n\n* For each StudyInstanceUID, number of images are from 1 to 9.","metadata":{}},{"cell_type":"code","source":"# How many images for each StudyInstanceUID?\nimage_level_study_id_value_counts = \\\n    image_level_df['StudyInstanceUID'].value_counts()\n\nimage_level_study_id_value_counts.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:39.711150Z","iopub.execute_input":"2021-06-08T10:47:39.711423Z","iopub.status.idle":"2021-06-08T10:47:39.723478Z","shell.execute_reply.started":"2021-06-08T10:47:39.711397Z","shell.execute_reply":"2021-06-08T10:47:39.722789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----\n## Image Predictions and Study Labels\n\nFor each study:\n\n* No 'opacity' and only 'none':\n    * Almost 'Negative'\n\n| 'opacity' | 'none' | Negative | Typical | Indeterminate | Atypical |\n|:---------:|:------:|:--------:|:-------:|:-------------:|:--------:|\n|     0     |  >= 1  |   1676   |    1    |       0       |    83    |\n\n* Some 'opacity' and zero or more 'none':\n    * NO 'Negative'\n\n| 'opacity' | 'none' | Negative | Typical | Indeterminate | Atypical |\n|:---------:|:------:|:--------:|:-------:|:-------------:|:--------:|\n|    >= 1   |   0    |    0     |  2724   |    1007       |    386   |\n|    >= 1   |  >= 1  |    0     |   130   |      42       |      5   |","metadata":{}},{"cell_type":"code","source":"# Append StudyInstanceUID to image level predictions.\nlabel_pred_study_id_df = \\\n    pd.concat([\n        image_level_label_pred_df,\n        image_level_df['StudyInstanceUID']],\n        axis=1)\n\nlabel_pred_study_id_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:39.724667Z","iopub.execute_input":"2021-06-08T10:47:39.725177Z","iopub.status.idle":"2021-06-08T10:47:39.743432Z","shell.execute_reply.started":"2021-06-08T10:47:39.725145Z","shell.execute_reply":"2021-06-08T10:47:39.742473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_0_or_ge_1(value):\n    return \"0\" if value == 0 else \">= 1\"","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:39.744755Z","iopub.execute_input":"2021-06-08T10:47:39.745045Z","iopub.status.idle":"2021-06-08T10:47:39.749594Z","shell.execute_reply.started":"2021-06-08T10:47:39.745018Z","shell.execute_reply":"2021-06-08T10:47:39.748431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For each study, check image prediction sum is \"0\" or \">= 1\".\ndef get_prediction_counts_for(study_id):\n    study_uid = study_id.replace(\"_study\", \"\")\n    study_uid_mask = \\\n        (label_pred_study_id_df['StudyInstanceUID'] == study_uid)\n    label_pred_for_study_df = \\\n        label_pred_study_id_df[ study_uid_mask ]\n    image_pred_count_for_study_df = \\\n        label_pred_for_study_df[ ['opacity', 'none'] ].sum()\n    opacity_value = get_0_or_ge_1(\n        image_pred_count_for_study_df['opacity'])\n    none_value = get_0_or_ge_1(\n        image_pred_count_for_study_df['none'])\n    return pd.Series({\n        'opacity': opacity_value, 'none' : none_value})\n\nstudy_level_image_pred_count_df = study_level_df['id'].apply(\n    lambda study_id: get_prediction_counts_for(study_id))\n\nstudy_level_image_pred_count_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:39.750977Z","iopub.execute_input":"2021-06-08T10:47:39.751493Z","iopub.status.idle":"2021-06-08T10:47:57.806295Z","shell.execute_reply.started":"2021-06-08T10:47:39.751448Z","shell.execute_reply":"2021-06-08T10:47:57.805274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many for image predictions and study labels?\nstudy_level_labels_df.columns = [\n    'Negative', 'Typical', 'Indeterminate', 'Atypical' ]\nimage_pred_count_study_labels_pd = pd.concat(\n    [study_level_image_pred_count_df, study_level_labels_df],\n    axis=1)\n\nimage_pred_count_study_labels_pd.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:57.807757Z","iopub.execute_input":"2021-06-08T10:47:57.808157Z","iopub.status.idle":"2021-06-08T10:47:57.826689Z","shell.execute_reply.started":"2021-06-08T10:47:57.808116Z","shell.execute_reply":"2021-06-08T10:47:57.825519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----\n## DICOM Series Number\n\nMake sure the topic regarding Series Number described [here](https://www.kaggle.com/c/siim-covid19-detection/discussion/243273).\n\nIn this discussion:\n> **Whichever image has the lowest SeriesNumber in the study is the one that you will need to predict bounding boxes on**.\n\nIf the lowest SerialNumber image has:\n* NO opacity,\n    * most of the other images have NO opacity.\n    * 8 images have opacity.\n* one or more opacity, ALL the other images have NO opacity.\n\n| lowest SeriesNumber opacity | the other SeriesNumber opacity | count |\n|:---------------------------:|:------------------------------:|:-----:|\n|                0            |                0               |  1760 |\n|                             |               >= 1             |     8 |\n|               >= 1          |                0               |  4286 |\n|                             |               >= 1             |     0 |\n","metadata":{}},{"cell_type":"code","source":"def get_image_path(image_level_row):\n    image_id = image_level_row['id'].replace('_image', '')\n    study_id = image_level_row['StudyInstanceUID']\n    image_path_pattern = os.path.join(\n        siim_covid19_dir, 'train', study_id, '*', image_id + \".dcm\")\n    image_path_list = glob.glob(image_path_pattern)\n    assert len(image_path_list) == 1\n    return image_path_list[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:57.831060Z","iopub.execute_input":"2021-06-08T10:47:57.831383Z","iopub.status.idle":"2021-06-08T10:47:57.836913Z","shell.execute_reply.started":"2021-06-08T10:47:57.831350Z","shell.execute_reply":"2021-06-08T10:47:57.835889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_series_number(image_level_row):\n    image_path = get_image_path(image_level_row)\n    # https://pydicom.github.io/pydicom/stable/reference/generated/pydicom.filereader.dcmread.html#pydicom.filereader.dcmread\n    # stop_before_pixels=True: to read element information only.\n    dicom = pydicom.filereader.dcmread(\n        image_path, stop_before_pixels=True)\n    if dicom.SeriesNumber is None:\n        series_number = -1   # Some DICOM file doesn't have the number...\n    else:\n        series_number = int(dicom.SeriesNumber)\n    return series_number","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:57.838541Z","iopub.execute_input":"2021-06-08T10:47:57.839038Z","iopub.status.idle":"2021-06-08T10:47:57.849938Z","shell.execute_reply.started":"2021-06-08T10:47:57.838985Z","shell.execute_reply":"2021-06-08T10:47:57.848911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_series_number_row(image_level_row):\n    series_number = get_series_number(image_level_row)\n    return pd.Series({\n        \"id\": image_level_row['id'],\n        'StudyInstanceUID': image_level_row['StudyInstanceUID'],\n        'SeriesNumber': series_number,\n    })","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:57.851169Z","iopub.execute_input":"2021-06-08T10:47:57.851595Z","iopub.status.idle":"2021-06-08T10:47:57.868349Z","shell.execute_reply.started":"2021-06-08T10:47:57.851565Z","shell.execute_reply":"2021-06-08T10:47:57.867555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_number_df = image_level_df.apply(make_series_number_row, axis=1)\nseries_number_df = pd.concat([\n    series_number_df, image_level_label_pred_df], axis=1)\n\nseries_number_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:47:57.869435Z","iopub.execute_input":"2021-06-08T10:47:57.869898Z","iopub.status.idle":"2021-06-08T10:49:59.496950Z","shell.execute_reply.started":"2021-06-08T10:47:57.869852Z","shell.execute_reply":"2021-06-08T10:49:59.495559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the study id and number of images for the study\nseries_number_grp = series_number_df.groupby(['StudyInstanceUID'])\nseries_number_grp_size = series_number_grp.size()\n\nseries_number_grp_size.sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:49:59.499173Z","iopub.execute_input":"2021-06-08T10:49:59.499636Z","iopub.status.idle":"2021-06-08T10:49:59.520182Z","shell.execute_reply.started":"2021-06-08T10:49:59.499573Z","shell.execute_reply":"2021-06-08T10:49:59.518999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For the study '0fd2db233deb', which has 9 images,\n# an image with lowest Series Number of '1' has an opacity bbox.\n# No opacity bboxes for the other images.\nseries_number_grp.get_group('0fd2db233deb').sort_values('SeriesNumber')","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:49:59.521894Z","iopub.execute_input":"2021-06-08T10:49:59.522388Z","iopub.status.idle":"2021-06-08T10:49:59.555209Z","shell.execute_reply.started":"2021-06-08T10:49:59.522345Z","shell.execute_reply":"2021-06-08T10:49:59.553991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The same for the study 'a7335b2f9815'.\nseries_number_grp.get_group('a7335b2f9815').sort_values('SeriesNumber')","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:49:59.556740Z","iopub.execute_input":"2021-06-08T10:49:59.557140Z","iopub.status.idle":"2021-06-08T10:49:59.573970Z","shell.execute_reply.started":"2021-06-08T10:49:59.557097Z","shell.execute_reply":"2021-06-08T10:49:59.572816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_series_number_group(ser_num_grp_df):\n    ser_num_grp_df = ser_num_grp_df.sort_values('SeriesNumber')\n    lowest = ser_num_grp_df.iloc[ 0, : ]\n    other = ser_num_grp_df.iloc[ 1: , : ]\n    return pd.Series({\n        'lowest_opacity': get_0_or_ge_1(lowest['opacity']),\n        'lowest_none': get_0_or_ge_1(lowest['none']),\n        'other_opacity': get_0_or_ge_1(other['opacity'].sum()),\n        'other_none': get_0_or_ge_1(other['none'].sum()) })","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:49:59.575623Z","iopub.execute_input":"2021-06-08T10:49:59.576058Z","iopub.status.idle":"2021-06-08T10:49:59.583383Z","shell.execute_reply.started":"2021-06-08T10:49:59.576014Z","shell.execute_reply":"2021-06-08T10:49:59.582521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_number_pred_count_df = \\\n    series_number_grp.apply(process_series_number_group)\n\nseries_number_pred_count_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:49:59.584492Z","iopub.execute_input":"2021-06-08T10:49:59.584965Z","iopub.status.idle":"2021-06-08T10:50:08.581446Z","shell.execute_reply.started":"2021-06-08T10:49:59.584924Z","shell.execute_reply":"2021-06-08T10:50:08.580324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_number_pred_count_df.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:50:08.582969Z","iopub.execute_input":"2021-06-08T10:50:08.583357Z","iopub.status.idle":"2021-06-08T10:50:08.600750Z","shell.execute_reply.started":"2021-06-08T10:50:08.583316Z","shell.execute_reply":"2021-06-08T10:50:08.599740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ser_num_pred_count_other_opacity_ge_1_df = \\\n    series_number_pred_count_df[\n        series_number_pred_count_df['other_opacity'] == \">= 1\" ]\n\nser_num_pred_count_other_opacity_ge_1_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:50:08.602343Z","iopub.execute_input":"2021-06-08T10:50:08.602821Z","iopub.status.idle":"2021-06-08T10:50:08.618853Z","shell.execute_reply.started":"2021-06-08T10:50:08.602778Z","shell.execute_reply":"2021-06-08T10:50:08.617928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for study_id in ser_num_pred_count_other_opacity_ge_1_df.index:\n    other_opacity_ge_1_study_df = series_number_grp.get_group(study_id)\n    print(other_opacity_ge_1_study_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:50:08.620152Z","iopub.execute_input":"2021-06-08T10:50:08.620454Z","iopub.status.idle":"2021-06-08T10:50:08.654330Z","shell.execute_reply.started":"2021-06-08T10:50:08.620427Z","shell.execute_reply":"2021-06-08T10:50:08.653417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----\n## BBoxes\n\nMajority pattern is 2 opacity bboxes for each lung.","metadata":{}},{"cell_type":"code","source":"def get_xray_size(image_level_row):\n    image_path = get_image_path(image_level_row)\n    dicom = pydicom.filereader.dcmread(\n        image_path, stop_before_pixels=True)\n    return int(dicom.Rows), int(dicom.Columns)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:50:08.655727Z","iopub.execute_input":"2021-06-08T10:50:08.656103Z","iopub.status.idle":"2021-06-08T10:50:08.661290Z","shell.execute_reply.started":"2021-06-08T10:50:08.656061Z","shell.execute_reply":"2021-06-08T10:50:08.660293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_opacity_count(image_level_row):\n    label_df = make_image_level_label_df(image_level_row['label'])\n    opacity_mask = label_df['prediction'] == \"opacity\"\n    opacity_count = sum(opacity_mask)\n    return opacity_count","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:50:08.662749Z","iopub.execute_input":"2021-06-08T10:50:08.663323Z","iopub.status.idle":"2021-06-08T10:50:08.673597Z","shell.execute_reply.started":"2021-06-08T10:50:08.663280Z","shell.execute_reply":"2021-06-08T10:50:08.672851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_image_level_bbox_row(image_level_row):\n    opacity_count = get_opacity_count(image_level_row)\n    height, width = get_xray_size(image_level_row)\n    return pd.Series({\n        'opacity_count': opacity_count,\n        'label': image_level_row['label'],\n        'height': height,\n        'width': width })\n\nimage_level_bbox_df = \\\n    image_level_df.apply(make_image_level_bbox_row, axis=1)\n\nimage_level_bbox_df","metadata":{"execution":{"iopub.status.busy":"2021-06-08T10:51:17.840751Z","iopub.execute_input":"2021-06-08T10:51:17.841135Z","iopub.status.idle":"2021-06-08T10:52:19.284437Z","shell.execute_reply.started":"2021-06-08T10:51:17.841096Z","shell.execute_reply":"2021-06-08T10:52:19.283590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opacity_mask = image_level_bbox_df['opacity_count'] > 0\nopacity_count = image_level_bbox_df.loc[opacity_mask, 'opacity_count'] \\\n    .value_counts() \\\n    .sort_index()\n\nopacity_count","metadata":{"execution":{"iopub.status.busy":"2021-06-08T11:18:49.054347Z","iopub.execute_input":"2021-06-08T11:18:49.054981Z","iopub.status.idle":"2021-06-08T11:18:49.065927Z","shell.execute_reply.started":"2021-06-08T11:18:49.054931Z","shell.execute_reply":"2021-06-08T11:18:49.064822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(opacity_count.index, opacity_count.values)\nplt.title('Opacity Counts for Each Image')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T11:19:04.244578Z","iopub.execute_input":"2021-06-08T11:19:04.245109Z","iopub.status.idle":"2021-06-08T11:19:04.405763Z","shell.execute_reply.started":"2021-06-08T11:19:04.245076Z","shell.execute_reply":"2021-06-08T11:19:04.404835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lefts = []\ntops = []\nrights = []\nbottoms = []\nfor _, bbox_row in image_level_bbox_df.iterrows():\n    label = bbox_row['label']\n    label_df = make_image_level_label_df(label)\n    for _, bbox in label_df.iterrows():\n        if bbox['prediction'] == 'none':\n            continue\n        left, top, right, bottom = \\\n            bbox[['left', 'top', 'right', 'bottom']]\n        image_width = bbox_row['width']\n        image_height = bbox_row['height']\n        lefts.append(left / image_width)\n        tops.append(top / image_height)\n        rights.append(right / image_width)\n        bottoms.append(bottom / image_height)\n\nlen(lefts)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T11:19:19.962724Z","iopub.execute_input":"2021-06-08T11:19:19.963223Z","iopub.status.idle":"2021-06-08T11:19:44.017273Z","shell.execute_reply.started":"2021-06-08T11:19:19.963190Z","shell.execute_reply":"2021-06-08T11:19:44.016253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 9))\nplt.scatter(lefts, tops, alpha=0.4, label=\"top-left\")\nplt.scatter(rights, bottoms, alpha=0.4, label=\"bottom-right\")\nplt.title(\"Opacity BBox Positions\")\nplt.gca().invert_yaxis()\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T11:19:59.073193Z","iopub.execute_input":"2021-06-08T11:19:59.073571Z","iopub.status.idle":"2021-06-08T11:19:59.723861Z","shell.execute_reply.started":"2021-06-08T11:19:59.073530Z","shell.execute_reply":"2021-06-08T11:19:59.722603Z"},"trusted":true},"execution_count":null,"outputs":[]}]}