{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-01T13:15:32.138828Z","iopub.execute_input":"2021-07-01T13:15:32.139217Z","iopub.status.idle":"2021-07-01T13:15:32.143514Z","shell.execute_reply.started":"2021-07-01T13:15:32.139183Z","shell.execute_reply":"2021-07-01T13:15:32.142207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import pydicom as dicom # dcm file\nimport matplotlib.pylab as plt # plot\nfrom matplotlib import patches # bounding box\nimport numpy as np # linear algebra\nimport pandas as pd # csv\nimport os # directory, folder, file\nimport threading\nprint(\"pydicom, matplotlib.pylab, pydicom, os, pandas, and threading successfully imported\")","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.147548Z","iopub.execute_input":"2021-07-01T13:15:32.147991Z","iopub.status.idle":"2021-07-01T13:15:32.158539Z","shell.execute_reply.started":"2021-07-01T13:15:32.147924Z","shell.execute_reply":"2021-07-01T13:15:32.157604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define folder/file paths","metadata":{}},{"cell_type":"code","source":"train_path = '../input/siim-covid19-detection/train'\ntest_path = '../input/siim-covid19-detection/test'\nimage_csv_path = '../input/siim-covid19-detection/train_image_level.csv'\nstudy_csv_path = '../input/siim-covid19-detection/train_study_level.csv'\nprint(\"Successfully assign paths to variables\")","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.160090Z","iopub.execute_input":"2021-07-01T13:15:32.160353Z","iopub.status.idle":"2021-07-01T13:15:32.176054Z","shell.execute_reply.started":"2021-07-01T13:15:32.160327Z","shell.execute_reply":"2021-07-01T13:15:32.175039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read image and study leve csv files and convert them to Pandas' DataFrame","metadata":{}},{"cell_type":"code","source":"study_df = pd.read_csv(study_csv_path)\nimage_df = pd.read_csv(image_csv_path)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.177514Z","iopub.execute_input":"2021-07-01T13:15:32.177838Z","iopub.status.idle":"2021-07-01T13:15:32.221507Z","shell.execute_reply.started":"2021-07-01T13:15:32.177806Z","shell.execute_reply":"2021-07-01T13:15:32.220622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display a patient's chest scan (utility function)","metadata":{}},{"cell_type":"code","source":"def displayChestScan(id):\n    ret = 0\n    for dirname, dirs, filenames in os.walk(train_path):\n        uid = dirname.split('/')[-2]\n        if (uid == id):\n            for filename in filenames:\n                try:\n                    image_path = os.path.join(dirname, filename)\n                    ds = dicom.dcmread(image_path)\n                    fig, img = plt.subplots(1, 1)\n                    img.imshow(ds.pixel_array)\n                    rect = patches.Rectangle((10, 10), 100, 100, \n                                                linewidth = 2,\n                                                edgecolor = 'r',\n                                                facecolor = 'none')\n                    img.add_patch(rect)\n                    plt.show()\n                except Exception as e:\n                    print('[Error][displayChestScan]', e)\n                    plt.close()\n                    ret = -1\n            break\n    \n    return ret","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.222863Z","iopub.execute_input":"2021-07-01T13:15:32.223125Z","iopub.status.idle":"2021-07-01T13:15:32.229956Z","shell.execute_reply.started":"2021-07-01T13:15:32.223100Z","shell.execute_reply":"2021-07-01T13:15:32.229092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display chest scan's bounding boxes and label (utility function)","metadata":{}},{"cell_type":"code","source":"def displayChestScanInfo(id):\n    for idx, row in image_df.iterrows():\n        if (row['StudyInstanceUID'] == id):\n            print(row)\n            break\n    \n    return row","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.231133Z","iopub.execute_input":"2021-07-01T13:15:32.231369Z","iopub.status.idle":"2021-07-01T13:15:32.241744Z","shell.execute_reply.started":"2021-07-01T13:15:32.231345Z","shell.execute_reply":"2021-07-01T13:15:32.240776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display chest scan by a user-given path (utility function)","metadata":{}},{"cell_type":"code","source":"def displayChestScanByPath(path):\n    ds = dicom.dcmread(path)\n    plt.imshow(ds.pixel_array)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.242934Z","iopub.execute_input":"2021-07-01T13:15:32.243310Z","iopub.status.idle":"2021-07-01T13:15:32.252430Z","shell.execute_reply.started":"2021-07-01T13:15:32.243282Z","shell.execute_reply":"2021-07-01T13:15:32.251475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Negative for Pneumonia","metadata":{}},{"cell_type":"code","source":"error_log = {}\nfor _, row in study_df.iterrows():\n    other_res = {row[key] for idx, key in enumerate(row.keys()) if (idx != 0 and idx != 1)} # 0: id, 1: Negative for Pneumonia\n    id = row['id'].split('_')[0]\n    neg_for_pnu = row['Negative for Pneumonia']\n    if (neg_for_pnu and 1 not in other_res):\n        ret = displayChestScan(id)\n        displayChestScanInfo(id)\n        \n        if (ret is 0):\n            break","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:32.253606Z","iopub.execute_input":"2021-07-01T13:15:32.253886Z","iopub.status.idle":"2021-07-01T13:15:34.446788Z","shell.execute_reply.started":"2021-07-01T13:15:32.253861Z","shell.execute_reply":"2021-07-01T13:15:34.445744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Typical Appearance","metadata":{}},{"cell_type":"code","source":"for _, row in study_df.iterrows():\n    other_res = {row[key] for idx, key in enumerate(row.keys()) if (idx != 0 and idx != 2)} # 0: id, 2: Typical Apperance\n    id = row['id'].split('_')[0]\n    typ_appr = row['Typical Appearance']\n    if (typ_appr and 1 not in other_res):\n        ret = displayChestScan(id)\n        displayChestScanInfo(id)\n        \n        if (ret is 0):\n            break","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:34.448049Z","iopub.execute_input":"2021-07-01T13:15:34.448331Z","iopub.status.idle":"2021-07-01T13:15:41.651818Z","shell.execute_reply.started":"2021-07-01T13:15:34.448304Z","shell.execute_reply":"2021-07-01T13:15:41.650758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Indeterminate Appearance","metadata":{}},{"cell_type":"code","source":"for _, row in study_df.iterrows():\n    other_res = {row[key] for idx, key in enumerate(row.keys()) if (idx != 0 and idx != 3)} # 0: id, 3: Indeterminate Appearance\n    id = row['id'].split('_')[0]\n    ind_appr = row['Indeterminate Appearance']\n    if (ind_appr and 1 not in other_res):\n        ret = displayChestScan(id)\n        displayChestScanInfo(id)\n        \n        if (ret is 0):\n            break","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:41.655469Z","iopub.execute_input":"2021-07-01T13:15:41.655755Z","iopub.status.idle":"2021-07-01T13:15:46.897692Z","shell.execute_reply.started":"2021-07-01T13:15:41.655724Z","shell.execute_reply":"2021-07-01T13:15:46.896738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Atypical Appearance","metadata":{}},{"cell_type":"code","source":"for _, row in study_df.iterrows():\n    other_res = {row[key] for idx, key in enumerate(row.keys()) if (idx != 0 and idx != 4)} # 0: id, 3: Indeterminate Appearance\n    id = row['id'].split('_')[0]\n    ind_appr = row['Atypical Appearance']\n    if (ind_appr and 1 not in other_res):\n        ret = displayChestScan(id)\n        displayChestScanInfo(id)\n        \n        if (ret is 0):\n            break","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:15:46.899164Z","iopub.execute_input":"2021-07-01T13:15:46.899758Z","iopub.status.idle":"2021-07-01T13:15:50.491160Z","shell.execute_reply.started":"2021-07-01T13:15:46.899704Z","shell.execute_reply":"2021-07-01T13:15:50.490036Z"},"trusted":true},"execution_count":null,"outputs":[]}]}