{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-04T08:56:52.276959Z","iopub.execute_input":"2021-09-04T08:56:52.277737Z","iopub.status.idle":"2021-09-04T08:57:41.517978Z","shell.execute_reply.started":"2021-09-04T08:56:52.277618Z","shell.execute_reply":"2021-09-04T08:57:41.514287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Import Libraries\nimport glob, pylab, pandas as pd\nimport pydicom, numpy as np","metadata":{"execution":{"iopub.status.busy":"2021-09-04T08:58:32.638429Z","iopub.execute_input":"2021-09-04T08:58:32.63883Z","iopub.status.idle":"2021-09-04T08:58:32.899697Z","shell.execute_reply.started":"2021-09-04T08:58:32.638791Z","shell.execute_reply":"2021-09-04T08:58:32.898483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport glob\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom matplotlib.patches import Rectangle\nfrom matplotlib import cm\nimport os\nimport seaborn as sns\nfrom skimage import measure\nimport keras\nimport tensorflow as tf\nimport cv2\n%matplotlib inline\n\n#setting up for customized printing\nfrom IPython.display import Markdown, display\nfrom IPython.display import HTML\ndef printmd(string, color=None):\n    colorstr = \"<span style='color:{}'>{}</span>\".format(color, string)\n    display(Markdown(colorstr))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T08:58:35.117356Z","iopub.execute_input":"2021-09-04T08:58:35.117766Z","iopub.status.idle":"2021-09-04T08:58:42.478085Z","shell.execute_reply.started":"2021-09-04T08:58:35.117734Z","shell.execute_reply":"2021-09-04T08:58:42.477058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def random_color():\n    rgbl=[1,0,0]\n    np.random.shuffle(rgbl)\n    return tuple(rgbl)  \n\ndef distplot(data, figRows,figCols,xSize, ySize, features, colors):\n    f, axes = plt.subplots(figRows, figCols, figsize=(xSize, ySize))\n    \n    features = np.array(features).reshape(figRows, figCols)\n    colors = np.array(colors).reshape(figRows, figCols)\n    \n    for row in range(figRows):\n        for col in range(figCols):\n            if (figRows == 1 and figCols == 1) :\n                axesplt = axes\n            elif (figRows == 1 and figCols > 1) :\n                axesplt = axes[col]\n            elif (figRows > 1 and figCols == 1) :\n                axesplt = axes[row]\n            else:\n                axesplt = axes[row][col]\n            plot = sns.distplot(data[features[row][col]], color=colors[row][col], ax=axesplt, kde=True, hist_kws={\"edgecolor\":\"k\"})\n            plot.set_xlabel(features[row][col],fontsize=20)\n            \ndef boxplot(data, figRows,figCols,xSize, ySize, features, colors, hue=None, orient='h'):\n    f, axes = plt.subplots(figRows, figCols, figsize=(xSize, ySize))\n    \n    features = np.array(features).reshape(figRows, figCols)\n    colors = np.array(colors).reshape(figRows, figCols)\n    \n    for row in range(figRows):\n        for col in range(figCols):\n            if (figRows == 1 and figCols == 1) :\n                axesplt = axes\n            elif (figRows == 1 and figCols > 1) :\n                axesplt = axes[col]\n            elif (figRows > 1 and figCols == 1) :\n                axesplt = axes[row]\n            else:\n                axesplt = axes[row][col]\n            plot = sns.boxplot(features[row][col], data= data, color=colors[row][col], ax=axesplt, orient=orient, hue=hue)\n            plot.set_xlabel(features[row][col],fontsize=20)\n            \ndef countplot(data, figRows,figCols,xSize, ySize, features, colors=None,palette=None,hue=None, orient=None, rotation=90):\n    f, axes = plt.subplots(figRows, figCols, figsize=(xSize, ySize))\n    \n    features = np.array(features).reshape(figRows, figCols)\n    if(colors is not None):\n        colors = np.array(colors).reshape(figRows, figCols)\n    if(palette is not None):\n        palette = np.array(palette).reshape(figRows, figCols)\n    \n    for row in range(figRows):\n        for col in range(figCols):\n            if (figRows == 1 and figCols == 1) :\n                axesplt = axes\n            elif (figRows == 1 and figCols > 1) :\n                axesplt = axes[col]\n            elif (figRows > 1 and figCols == 1) :\n                axesplt = axes[row]\n            else:\n                axesplt = axes[row][col]\n                \n            if(colors is None):\n                plot = sns.countplot(features[row][col], data=data, palette=palette[row][col], ax=axesplt, orient=orient, hue=hue)\n            elif(palette is None):\n                plot = sns.countplot(features[row][col], data=data, color=colors[row][col], ax=axesplt, orient=orient, hue=hue)\n            plot.set_title(features[row][col],fontsize=20)\n            plot.set_xlabel(None)\n            plot.set_xticklabels(rotation=rotation, labels=plot.get_xticklabels(),fontweight='demibold',fontsize='large')\n            \ndef point_box_bar_plot(data, row, col, target, figRow, figCol, palette='rocket', fontsize='large', fontweight='demibold'):\n    sns.set(style=\"whitegrid\")\n    f, axes = plt.subplots(3, 1, figsize=(figRow, figCol))\n    pplot=sns.pointplot(row,col, data=data, ax=axes[0], linestyles=['--'])\n    pplot.set_xlabel(None)\n    pplot.set_xticklabels(labels=pplot.get_xticklabels(),fontweight=fontweight,fontsize=fontsize)\n    bxplot=sns.boxplot(row,col, data=data, hue=target, ax=axes[1],palette='viridis')\n    bxplot.set_xlabel(None)\n    bxplot.set_xticklabels(labels=bxplot.get_xticklabels(),fontweight=fontweight,fontsize=fontsize)\n    bplot=sns.barplot(row,col, data=data, hue=target, ax=axes[2],palette=palette)\n    bplot.set_xlabel(row,fontsize=20)\n    bplot.set_xticklabels(labels=bplot.get_xticklabels(),fontweight=fontweight,fontsize=fontsize)            \ndef catdist(data, cols):\n    dfs = []\n    for col in cols:\n        colData = pd.DataFrame(data[col].value_counts(), columns=[col])\n        colData['%'] = round((colData[col]/colData[col].sum())*100,2)\n        dfs.append(colData)\n        display(colData)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T08:58:42.480083Z","iopub.execute_input":"2021-09-04T08:58:42.480551Z","iopub.status.idle":"2021-09-04T08:58:42.511799Z","shell.execute_reply.started":"2021-09-04T08:58:42.480504Z","shell.execute_reply":"2021-09-04T08:58:42.51064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Setup Data","metadata":{"execution":{"iopub.status.busy":"2021-09-04T08:58:42.514195Z","iopub.execute_input":"2021-09-04T08:58:42.514681Z","iopub.status.idle":"2021-09-04T08:58:42.531024Z","shell.execute_reply.started":"2021-09-04T08:58:42.514636Z","shell.execute_reply":"2021-09-04T08:58:42.529869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def formatMetadataString(val):\n    return str(val).split(':')[1].replace('\\'', '')\n\ndataDir = '/kaggle/input/rsna-pneumonia-detection-challenge'\ntrainDataDir = 'stage_2_train_images'\ntestDataDir = 'stage_2_test_images'\ntrainfiles = glob.glob(os.path.join(dataDir, trainDataDir, \"*.dcm\"))\n\nprintmd('**Reading the metadata from dicom training files**', color='Blue')\n\nimage_data = []\nfor f in trainfiles:    \n    ds = pydicom.dcmread(f, stop_before_pixels=True)\n    patientId = formatMetadataString(ds['PatientID'])\n    age = formatMetadataString(ds['PatientAge'])\n    gender = formatMetadataString(ds['PatientSex'])\n    viewPos = formatMetadataString(ds['ViewPosition'])    \n    image_data.append([patientId, int(age), gender, viewPos])\n    \nimage_metadata_df = pd.DataFrame(image_data, columns=['patientId', 'Age', 'Gender','ViewPosition'])        \nimage_metadata_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:02:16.90609Z","iopub.execute_input":"2021-09-04T09:02:16.906525Z","iopub.status.idle":"2021-09-04T09:03:06.898197Z","shell.execute_reply.started":"2021-09-04T09:02:16.906482Z","shell.execute_reply":"2021-09-04T09:03:06.897172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Storing the DICOM files as a dictionary under trainDicomFiles such that the keys are the patient\\\nids and the values are the path of the files corresponding to the image.')\ntrainDicomFiles = {}\ntestDicomFiles = {}\n\nfor root, dirs, files in os.walk(os.path.join(dataDir, trainDataDir)):\n    for fileName in files:\n        if '.dcm' in fileName.lower():\n            trainDicomFiles[fileName.split(sep='.')[0]] = os.path.join(dataDir, trainDataDir, fileName)     \n        \nprint('\\nGenerating the data frame for the training data labels.')\ntrainLabelDf = pd.read_csv(os.path.join(dataDir, 'stage_2_train_labels.csv'))\ntrainLabelDetailedDf = pd.read_csv(os.path.join(dataDir, 'stage_2_detailed_class_info.csv'))\n\n# Joining the class information from the detailed class info CSV to the train labels CSV.\ntrainLabelDf['class'] = trainLabelDetailedDf['class']\n\n# As the training labels file contain multiple rows for each patient corresponding to the number of bounding boxes present\n# in case of a person with lung opacity, we can group the labels by patient id to find number of unique patients in \n# the label file.\n\nprint('\\nGenerating the dictionary for test DICOM files under testDicomFiles.')\nfor root, dirs, files in os.walk(os.path.join(dataDir, testDataDir)):\n    for fileName in files:\n        if '.dcm' in fileName.lower():\n            testDicomFiles[fileName.split(sep='.')[0]] = os.path.join(dataDir, testDataDir, fileName)\n\nprint('Total test data files: ', len(testDicomFiles))\n\n#print('\\nBelow is a sample of the content of training label data frame.')\n#display(trainLabelDf.head(10))\n\nprint('\\nAs the training label data frame may contain multiple records for a patient to represent the bounding box, we can \\\nstore the bounding box labels in a property called boundingBox that contains a array of box data stored as \\\n{x, y, width, height}')\n\nuniquePatientIds = np.unique(trainLabelDf.patientId)\ntrainLabelAndBoundingBoxDf = pd.DataFrame(columns=['patientId', 'target', 'class', 'boundingBox', 'num_of_opacities']\n                                          , index=uniquePatientIds)\n\npatientIdBasedGroups = trainLabelDf.groupby('patientId')\n#counter = 0\nfor patientId, group in patientIdBasedGroups:\n#     if counter > 20:\n#         break\n    trainLabelAndBoundingBoxDf.loc[patientId]['patientId'] = patientId\n    trainLabelAndBoundingBoxDf.loc[patientId]['num_of_opacities'] = 0\n    boundingBox = []\n    for index, groupItem in group.iterrows():\n        trainLabelAndBoundingBoxDf.loc[patientId]['target'] = groupItem['Target']\n        trainLabelAndBoundingBoxDf.loc[patientId]['class'] = groupItem['class']\n        \n        if(not np.isnan(groupItem['x'])):\n            boundingBox.append((groupItem['x'], groupItem['y'], groupItem['width'], groupItem['height']))\n            trainLabelAndBoundingBoxDf.loc[patientId]['num_of_opacities'] += 1\n            \n    trainLabelAndBoundingBoxDf.loc[patientId]['boundingBox'] = boundingBox\n#     counter += 1\n\ndisplay(trainLabelAndBoundingBoxDf.head())\n\nprint('\\nShape of training labels with bounding box list: ', trainLabelAndBoundingBoxDf.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:04:31.297943Z","iopub.execute_input":"2021-09-04T09:04:31.299405Z","iopub.status.idle":"2021-09-04T09:05:28.669066Z","shell.execute_reply.started":"2021-09-04T09:04:31.299344Z","shell.execute_reply":"2021-09-04T09:05:28.667856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printmd('**Combining training data with labels and bounding boxes with the image metadata containing age, gender and view position attributes**', color='brown')\ntrainLabelAndBoundingBoxDf_2 = trainLabelAndBoundingBoxDf.reset_index().drop('index', axis=1)\npatient_df = image_metadata_df.join(trainLabelAndBoundingBoxDf_2, how='inner', rsuffix='_2')\npatient_df.drop('patientId_2', axis=1, inplace=True)\npatient_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:05:28.671302Z","iopub.execute_input":"2021-09-04T09:05:28.671737Z","iopub.status.idle":"2021-09-04T09:05:28.723005Z","shell.execute_reply.started":"2021-09-04T09:05:28.671691Z","shell.execute_reply":"2021-09-04T09:05:28.721875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:05:49.283203Z","iopub.execute_input":"2021-09-04T09:05:49.283618Z","iopub.status.idle":"2021-09-04T09:05:49.288962Z","shell.execute_reply.started":"2021-09-04T09:05:49.283585Z","shell.execute_reply":"2021-09-04T09:05:49.287976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Let's go ahead and take a look at the first labels CSV file first:","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\nprint(df.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:05:51.437175Z","iopub.execute_input":"2021-09-04T09:05:51.437752Z","iopub.status.idle":"2021-09-04T09:05:51.484037Z","shell.execute_reply.started":"2021-09-04T09:05:51.437717Z","shell.execute_reply":"2021-09-04T09:05:51.483203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:05:53.236935Z","iopub.execute_input":"2021-09-04T09:05:53.237617Z","iopub.status.idle":"2021-09-04T09:05:53.241696Z","shell.execute_reply.started":"2021-09-04T09:05:53.237563Z","shell.execute_reply":"2021-09-04T09:05:53.240354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, each row in the CSV file contains a patientId (one unique value per patient), a target (either 0 or 1 for absence or presence of pneumonia, respectively) and the corresponding abnormality bounding box defined by the upper-left hand corner (x, y) coordinate and its corresponding width and height. In this particular case, the patient does not have pneumonia and so the corresponding bounding box information is set to NaN.","metadata":{}},{"cell_type":"code","source":"print(df.iloc[4])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:05:55.83671Z","iopub.execute_input":"2021-09-04T09:05:55.837151Z","iopub.status.idle":"2021-09-04T09:05:55.843871Z","shell.execute_reply.started":"2021-09-04T09:05:55.837115Z","shell.execute_reply":"2021-09-04T09:05:55.842748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:05:57.122922Z","iopub.execute_input":"2021-09-04T09:05:57.123608Z","iopub.status.idle":"2021-09-04T09:05:57.127764Z","shell.execute_reply.started":"2021-09-04T09:05:57.123566Z","shell.execute_reply":"2021-09-04T09:05:57.126791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One important thing to keep in mind is that a given patientId may have multiple boxes if more than one area of pneumonia is detected (see below for example images).","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Medical images are stored in a special format known as DICOM files (*.dcm). They contain a combination of header metadata as well as underlying raw image arrays for pixel data. In Python, one popular library to access and manipulate DICOM files is the pydicom module. To use the pydicom library, first find the DICOM file for a given patientId by simply looking for the matching file in the stage_2_train_images/ folder, and the use the pydicom.read_file() method to load the data:","metadata":{}},{"cell_type":"code","source":"patientId = df['patientId'][0]\ndcm_file = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % patientId\ndcm_data = pydicom.read_file(dcm_file)\n\nprint(dcm_data)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:02.91723Z","iopub.execute_input":"2021-09-04T09:06:02.917624Z","iopub.status.idle":"2021-09-04T09:06:02.940376Z","shell.execute_reply.started":"2021-09-04T09:06:02.917592Z","shell.execute_reply":"2021-09-04T09:06:02.939163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the standard headers containing patient identifable information have been anonymized (removed) so we are left with a relatively sparse set of metadata. The primary field we will be accessing is the underlying pixel data as follows:","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"im = dcm_data.pixel_array\nprint(type(im))\nprint(im.dtype)\nprint(im.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:07.282192Z","iopub.execute_input":"2021-09-04T09:06:07.282743Z","iopub.status.idle":"2021-09-04T09:06:07.324825Z","shell.execute_reply.started":"2021-09-04T09:06:07.282699Z","shell.execute_reply":"2021-09-04T09:06:07.324061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Considerations:\n\nAs we can see here, the pixel array data is stored as a Numpy array, a powerful numeric Python library for handling and manipulating matrix data (among other things). In addition, it is apparent here that the original radiographs have been preprocessed for us as follows:\n\nThe relatively high dynamic range, high bit-depth original images have been rescaled to 8-bit encoding (256 grayscales). For the radiologists out there, this means that the images have been windowed and leveled already. In clinical practice, manipulating the image bit-depth is typically done manually by a radiologist to highlight certain disease processes. To visually assess the quality of the automated bit-depth downscaling and for considerations on potentially improving this baseline, consider consultation with a radiologist physician.\n\nThe relativley large original image matrices (typically acquired at >2000 x 2000) have been resized to the data-science friendly shape of 1024 x 1024. For the purposes of this challenge, the diagnosis of most pneumonia cases can typically be made at this resolution. To visually assess the feasibility of diagnosis at this resolution, and to determine the optimal resolution for pneumonia detection (oftentimes can be done at a resolution even smaller than 1024 x 1024), consider consultation with a radiogist physician.\n\n","metadata":{}},{"cell_type":"markdown","source":"Visualizing :\n\nTo take a look at this first DICOM image, let's use the pylab.imshow() method:","metadata":{}},{"cell_type":"code","source":"pylab.imshow(im, cmap=pylab.cm.gist_gray)\npylab.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:11.117048Z","iopub.execute_input":"2021-09-04T09:06:11.119404Z","iopub.status.idle":"2021-09-04T09:06:11.443492Z","shell.execute_reply.started":"2021-09-04T09:06:11.119352Z","shell.execute_reply":"2021-09-04T09:06:11.442115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Exploring the Data and Labels:\n\nAs alluded to above, any given patient may potentially have many boxes if there are several different suspicious areas of pneumonia. To collapse the current CSV file dataframe into a dictionary with unique entries, consider the following method:","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def parse_data(df):\n    \n    \n    # --- Define lambda to extract coords in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in df.iterrows():\n        # --- Initialize patient entry into parsed \n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                'dicom': '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n\n        # --- Add box if opacity is present\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:15.557275Z","iopub.execute_input":"2021-09-04T09:06:15.557655Z","iopub.status.idle":"2021-09-04T09:06:15.565364Z","shell.execute_reply.started":"2021-09-04T09:06:15.557623Z","shell.execute_reply":"2021-09-04T09:06:15.564238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parsed = parse_data(df)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:17.407368Z","iopub.execute_input":"2021-09-04T09:06:17.407869Z","iopub.status.idle":"2021-09-04T09:06:20.332003Z","shell.execute_reply.started":"2021-09-04T09:06:17.407812Z","shell.execute_reply":"2021-09-04T09:06:20.330995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we saw above, patient 00436515-870c-4b36-a041-de91049b9ab4 has pnuemonia so lets check our new parsed dict here to see the patients corresponding bounding boxes:","metadata":{}},{"cell_type":"code","source":"#00436515-870c-4b36-a041-de91049b9ab4\n\nprint(parsed['00436515-870c-4b36-a041-de91049b9ab4'])\n","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:22.356761Z","iopub.execute_input":"2021-09-04T09:06:22.35729Z","iopub.status.idle":"2021-09-04T09:06:22.362444Z","shell.execute_reply.started":"2021-09-04T09:06:22.357255Z","shell.execute_reply":"2021-09-04T09:06:22.36145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualizing Boxes:\n\nIn order to overlay color boxes on the original grayscale DICOM files, consider using the following methods (below, the main method draw() requires the method overlay_box()):","metadata":{}},{"cell_type":"code","source":"def draw(data):\n    \"\"\"\n    Method to draw single patient with bounding box(es) if present \n\n    \"\"\"\n    # --- Open DICOM file\n    d = pydicom.read_file(data['dicom'])\n    im = d.pixel_array\n\n    # --- Convert from single-channel grayscale to 3-channel RGB\n    im = np.stack([im] * 3, axis=2)\n\n    # --- Add boxes with random color if present\n    for box in data['boxes']:\n        rgb = np.floor(np.random.rand(3) * 256).astype('int')\n        im = overlay_box(im=im, box=box, rgb=rgb, stroke=6)\n\n    pylab.imshow(im, cmap=pylab.cm.gist_gray)\n    pylab.axis('off')\n\ndef overlay_box(im, box, rgb, stroke=1):\n    \"\"\"\n    Method to overlay single box on image\n\n    \"\"\"\n    # --- Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    # --- Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n\n    return im","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:24.15715Z","iopub.execute_input":"2021-09-04T09:06:24.157522Z","iopub.status.idle":"2021-09-04T09:06:24.169475Z","shell.execute_reply.started":"2021-09-04T09:06:24.157491Z","shell.execute_reply":"2021-09-04T09:06:24.167995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we saw above, patient 00436515-870c-4b36-a041-de91049b9ab4 has pnuemonia so let's take a look at the overlaid bounding boxes:","metadata":{}},{"cell_type":"code","source":"draw(parsed['00436515-870c-4b36-a041-de91049b9ab4'])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:26.116993Z","iopub.execute_input":"2021-09-04T09:06:26.117441Z","iopub.status.idle":"2021-09-04T09:06:26.400277Z","shell.execute_reply.started":"2021-09-04T09:06:26.117406Z","shell.execute_reply":"2021-09-04T09:06:26.399086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Perform Basic EDA","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print('Total training data files: ', len(trainDicomFiles))\nprint('Total rows in the labels file: ', trainLabelDf.shape[0])\nprint('Total unique patients in the labels file: ', trainLabelDf.groupby('patientId').count().shape[0])\nprint('Unique target values are:', np.unique(trainLabelDf['Target']).tolist())\nprint('Unique class values are: ', np.unique(trainLabelDf['class']).tolist())","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:29.636943Z","iopub.execute_input":"2021-09-04T09:06:29.637373Z","iopub.status.idle":"2021-09-04T09:06:29.738862Z","shell.execute_reply.started":"2021-09-04T09:06:29.63734Z","shell.execute_reply":"2021-09-04T09:06:29.737519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printmd('**We can check if the bounding boxes are present only for the patients with lung opacity.**', color='brown')\nrecordsWithLungOpacityAndBoundingBox = trainLabelAndBoundingBoxDf[(trainLabelAndBoundingBoxDf['boundingBox'].apply(lambda row: len(row) > 0)) & (trainLabelAndBoundingBoxDf['class'] == 'Lung Opacity')].shape[0]\nrecordsWithNoLungOpacityAndBoundingBox = trainLabelAndBoundingBoxDf[(trainLabelAndBoundingBoxDf['boundingBox'].apply(lambda row: len(row) > 0)) & (trainLabelAndBoundingBoxDf['class'] != 'Lung Opacity')].shape[0]\nprint('Record count with bounding box for patients with lung opacity: ', recordsWithLungOpacityAndBoundingBox)\nprint('Record count with bounding box for patients with no lung opacity: ', recordsWithNoLungOpacityAndBoundingBox)\nprint('We can see that bounding boxes are present for only patients with a lung opacity i.e. the patients having pneumonia.')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:31.121981Z","iopub.execute_input":"2021-09-04T09:06:31.122441Z","iopub.status.idle":"2021-09-04T09:06:31.171034Z","shell.execute_reply.started":"2021-09-04T09:06:31.122406Z","shell.execute_reply":"2021-09-04T09:06:31.169456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Analysis of DICOM file samples**","metadata":{}},{"cell_type":"code","source":"printmd('**Analyzing a sample with normal lung image.**', color='brown')\nsampleNormalLungPatient = trainLabelAndBoundingBoxDf[trainLabelAndBoundingBoxDf['class'] == 'Normal'].iloc[0]\nsampleNormalLungDicomFile = pydicom.read_file(trainDicomFiles[sampleNormalLungPatient.patientId])\n#display(sampleNormalLungDicomFile)\n\nprintmd('**Below is the image plot for the normal lung image stored in the DICOM file.**', color='brown')\nplt.imshow(sampleNormalLungDicomFile.pixel_array, cmap=cm.bone)\nplt.show()\n\nprintmd('**Below is the class target for the normal lung sample.**', color='brown')\nsampleNormalLungPatient","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:33.397106Z","iopub.execute_input":"2021-09-04T09:06:33.397558Z","iopub.status.idle":"2021-09-04T09:06:33.706363Z","shell.execute_reply.started":"2021-09-04T09:06:33.397523Z","shell.execute_reply":"2021-09-04T09:06:33.705368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printmd('**Analyzing a sample with lung opacity image.**', color='brown')\nsampleLungOpacityPatient = trainLabelAndBoundingBoxDf[trainLabelAndBoundingBoxDf['class'] == 'Lung Opacity'].iloc[0]\nsampleLungOpactiyDicomFile = pydicom.read_file(trainDicomFiles[sampleLungOpacityPatient.patientId])\n#display(sampleLungOpactiyDicomFile)\n\nprintmd('**Below is the image plot for the lung opacity image stored in the DICOM file.**', color='brown')\nfig = plt.figure()\nax = fig.add_subplot(111)\nax.imshow(sampleLungOpactiyDicomFile.pixel_array, cmap=cm.bone)\nfor box in sampleLungOpacityPatient['boundingBox']:\n    #print(box)\n    ax.add_patch(Rectangle((box[0], box[1]), box[2], box[3], linewidth=1, edgecolor=random_color(), facecolor='none'))\nplt.show()\n\nprintmd('**Below is the class target for the lung opacity sample.**', color='brown')\nsampleLungOpacityPatient","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:34.601923Z","iopub.execute_input":"2021-09-04T09:06:34.602303Z","iopub.status.idle":"2021-09-04T09:06:34.88961Z","shell.execute_reply.started":"2021-09-04T09:06:34.602269Z","shell.execute_reply":"2021-09-04T09:06:34.888865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"printmd('**Analyzing a sample with no lung opacity and not normal image.**', color='brown')\nsampleLungNotNormalPatient = trainLabelAndBoundingBoxDf[trainLabelAndBoundingBoxDf['class'] == 'No Lung Opacity / Not Normal'].iloc[0]\nsampleLungNotNormalDicomFile = pydicom.read_file(trainDicomFiles[sampleLungNotNormalPatient.patientId])\n#display(sampleLungNotNormalDicomFile)\n\nprintmd('**Below is the image plot for the no lung opacity and not normal image stored in the DICOM file.**', color='brown')\nplt.imshow(sampleLungNotNormalDicomFile.pixel_array, cmap=cm.bone)\nplt.show()\n\nprintmd('**Below is the class target for the no lung opacity and not normal image sample.**', color='brown')\nsampleLungNotNormalPatient","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:35.802217Z","iopub.execute_input":"2021-09-04T09:06:35.802781Z","iopub.status.idle":"2021-09-04T09:06:36.442682Z","shell.execute_reply.started":"2021-09-04T09:06:35.802738Z","shell.execute_reply":"2021-09-04T09:06:36.441402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dealing with missing values**","metadata":{}},{"cell_type":"code","source":"print('We can check the missing values in the train class files.')\n\nprint('Using the count function on the labels data frame, we can find if there exist any column with \"nan\" values.')\ndisplay(trainLabelAndBoundingBoxDf.count())\nprint('\\nAs we can see that all patientId, target and class does not contain any nan values as count is same as total number \\\nof patients.')\n\nprint('\\nWe can check the unique target and class values to determine if there is any ambiguous/missing data.')\nprint('Unique target values are:', np.unique(trainLabelAndBoundingBoxDf['target']).tolist())\nprint('Unique class values are: ', np.unique(trainLabelAndBoundingBoxDf['class']).tolist())\nprint('We can see there exist no invalid value for target and class data.')\n\nprint('\\nNow we can analyze the box coordinates for all the data points where Target is 1 (Pneumonia detected).')\ndisplay(trainLabelDf[trainLabelDf['Target'] == 1].describe())\n\nprint('\\nWe can see from statistics all the x, y coordinates contain valid values and none of x, y, width and height columns\\\n contain the min value as 0 indicating the values would be correctly recorded.')\n\nprint('\\nWe can check to see if all the patients within the train label data set has a corresponding DICOM image available.')\nfor labelRow in trainLabelAndBoundingBoxDf.iterrows():\n    if labelRow[1].patientId not in trainDicomFiles.keys():\n        print(f'{labelRow[1].patientId} is not present in the DICOM files.')\n\nprint('We can see that all the patients in labels file have a corresponding DICOM image file.')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:40.04188Z","iopub.execute_input":"2021-09-04T09:06:40.04227Z","iopub.status.idle":"2021-09-04T09:06:42.690697Z","shell.execute_reply.started":"2021-09-04T09:06:40.04224Z","shell.execute_reply":"2021-09-04T09:06:42.689566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distributions of Numeric Attribute - Age**","metadata":{}},{"cell_type":"code","source":"distplot(patient_df, 1,1, 12,5, \n         ['Age'], \n         ['red'])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:42.692271Z","iopub.execute_input":"2021-09-04T09:06:42.6926Z","iopub.status.idle":"2021-09-04T09:06:43.426322Z","shell.execute_reply.started":"2021-09-04T09:06:42.692568Z","shell.execute_reply":"2021-09-04T09:06:43.42494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n\nThere is a good distribution of people from all ages in the data. From the plot, it looks is normally distributed with majority of people around ages 40 to 60.","metadata":{}},{"cell_type":"markdown","source":"Distribution of Categorical Attributes","metadata":{}},{"cell_type":"code","source":"catdist(patient_df, ['Gender', 'class', 'target', 'ViewPosition'])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:47.67687Z","iopub.execute_input":"2021-09-04T09:06:47.677415Z","iopub.status.idle":"2021-09-04T09:06:47.76622Z","shell.execute_reply.started":"2021-09-04T09:06:47.677374Z","shell.execute_reply":"2021-09-04T09:06:47.765121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Bivariate Analysis**","metadata":{}},{"cell_type":"code","source":"printmd('**Attributes target - Age**', color='brown')\npoint_box_bar_plot(patient_df, 'target', 'Age', 'target', 15, 10, palette='winter')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:50.156959Z","iopub.execute_input":"2021-09-04T09:06:50.157343Z","iopub.status.idle":"2021-09-04T09:06:51.531443Z","shell.execute_reply.started":"2021-09-04T09:06:50.157311Z","shell.execute_reply":"2021-09-04T09:06:51.53043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n\n1. Average age of people who are normal and who are affected by pneumonia are the same at around 47.\n2. Distribution of people's age is quite similar for both Normal and Pneumonia affected.","metadata":{}},{"cell_type":"code","source":"printmd('**Attributes class - Age**', color='brown')\npoint_box_bar_plot(patient_df, 'class', 'Age', 'class', 15, 10, palette='winter')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:53.277122Z","iopub.execute_input":"2021-09-04T09:06:53.277565Z","iopub.status.idle":"2021-09-04T09:06:54.918039Z","shell.execute_reply.started":"2021-09-04T09:06:53.27753Z","shell.execute_reply":"2021-09-04T09:06:54.916921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Observations:\n\n1. Average age of people across different classes (Normal, Lung Opacity, No Lung Opacity/Not Normal) are same at around 47.\n2. Distribution of people's age is quite similar for class types.","metadata":{}},{"cell_type":"code","source":"printmd('**Distribution of data/images accross different classes.**', color='brown')\n\nfig, axes = plt.subplots(1)\ncol1plot = sns.countplot(x='class', data=patient_df)\nfor p in col1plot.patches:\n    col1plot.annotate(format(p.get_height(), '.2f'), (p.get_x() + p.get_width() / 2., p.get_height()), ha = 'center', va = 'center', xytext = (0, 10), textcoords = 'offset points')\n\nfig.set_figwidth(10)\nfig.set_figheight(8)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:57.517068Z","iopub.execute_input":"2021-09-04T09:06:57.517499Z","iopub.status.idle":"2021-09-04T09:06:57.751334Z","shell.execute_reply.started":"2021-09-04T09:06:57.517462Z","shell.execute_reply":"2021-09-04T09:06:57.750351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations\nThere is a good ratio of class distributions. \nDifferent class types are distributed as below: \n\n1. No Lung Opacity / Not Normal - 11821 - 44.30%\n2. Normal - 8851 - 33.17%\n3. Lung Opacity - 6012 - 22.53%","metadata":{}},{"cell_type":"code","source":"printmd('**Distribution of categorical attributes wrt target attribute**', color='brown')\ncountplot(patient_df, 2,2, 20,10, \n         ['target', 'Gender', 'ViewPosition', 'class'], \n         palette=['viridis', 'Dark2_r', 'winter', 'Set1'], hue='target', rotation=60)\n","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:06:59.996743Z","iopub.execute_input":"2021-09-04T09:06:59.997192Z","iopub.status.idle":"2021-09-04T09:07:00.967805Z","shell.execute_reply.started":"2021-09-04T09:06:59.997155Z","shell.execute_reply":"2021-09-04T09:07:00.967026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\n\npatient_df.hist(figsize=(20,30))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:02.516823Z","iopub.execute_input":"2021-09-04T09:07:02.517397Z","iopub.status.idle":"2021-09-04T09:07:02.97965Z","shell.execute_reply.started":"2021-09-04T09:07:02.517353Z","shell.execute_reply":"2021-09-04T09:07:02.978161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nsns.boxplot(x=\"Gender\", y=\"Age\", data=patient_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:03.916717Z","iopub.execute_input":"2021-09-04T09:07:03.917121Z","iopub.status.idle":"2021-09-04T09:07:04.128163Z","shell.execute_reply.started":"2021-09-04T09:07:03.917087Z","shell.execute_reply":"2021-09-04T09:07:04.127076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:05.237126Z","iopub.execute_input":"2021-09-04T09:07:05.237553Z","iopub.status.idle":"2021-09-04T09:07:05.259935Z","shell.execute_reply.started":"2021-09-04T09:07:05.23752Z","shell.execute_reply":"2021-09-04T09:07:05.258573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(patient_df['ViewPosition'],patient_df['Gender'] )","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:06.557436Z","iopub.execute_input":"2021-09-04T09:07:06.557928Z","iopub.status.idle":"2021-09-04T09:07:06.617817Z","shell.execute_reply.started":"2021-09-04T09:07:06.557893Z","shell.execute_reply":"2021-09-04T09:07:06.616507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(patient_df['patientId'],patient_df['Age'] )","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:07.677082Z","iopub.execute_input":"2021-09-04T09:07:07.677516Z","iopub.status.idle":"2021-09-04T09:07:08.51985Z","shell.execute_reply.started":"2021-09-04T09:07:07.677481Z","shell.execute_reply":"2021-09-04T09:07:08.518819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(patient_df['class'],patient_df['Age'] ) #Gender","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:09.717612Z","iopub.execute_input":"2021-09-04T09:07:09.718085Z","iopub.status.idle":"2021-09-04T09:07:09.803286Z","shell.execute_reply.started":"2021-09-04T09:07:09.718049Z","shell.execute_reply":"2021-09-04T09:07:09.801852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(patient_df['class'],patient_df['Gender'] ) ","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:11.557077Z","iopub.execute_input":"2021-09-04T09:07:11.557534Z","iopub.status.idle":"2021-09-04T09:07:11.599315Z","shell.execute_reply.started":"2021-09-04T09:07:11.557499Z","shell.execute_reply":"2021-09-04T09:07:11.597869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"class\", hue=\"Gender\", data=patient_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:13.196987Z","iopub.execute_input":"2021-09-04T09:07:13.197403Z","iopub.status.idle":"2021-09-04T09:07:13.510481Z","shell.execute_reply.started":"2021-09-04T09:07:13.197367Z","shell.execute_reply":"2021-09-04T09:07:13.509466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"Age\", hue=\"Gender\", data=patient_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:14.837187Z","iopub.execute_input":"2021-09-04T09:07:14.837579Z","iopub.status.idle":"2021-09-04T09:07:18.234995Z","shell.execute_reply.started":"2021-09-04T09:07:14.837537Z","shell.execute_reply":"2021-09-04T09:07:18.233873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"ViewPosition\", hue=\"Gender\", data=patient_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:18.236612Z","iopub.execute_input":"2021-09-04T09:07:18.236927Z","iopub.status.idle":"2021-09-04T09:07:18.506257Z","shell.execute_reply.started":"2021-09-04T09:07:18.236898Z","shell.execute_reply":"2021-09-04T09:07:18.505116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(patient_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:19.956585Z","iopub.execute_input":"2021-09-04T09:07:19.956971Z","iopub.status.idle":"2021-09-04T09:07:23.367787Z","shell.execute_reply.started":"2021-09-04T09:07:19.956938Z","shell.execute_reply":"2021-09-04T09:07:23.366445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df['Age'].std()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:48.557306Z","iopub.execute_input":"2021-09-04T09:07:48.557672Z","iopub.status.idle":"2021-09-04T09:07:48.564972Z","shell.execute_reply.started":"2021-09-04T09:07:48.55764Z","shell.execute_reply":"2021-09-04T09:07:48.563482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df['Age'].mean()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:49.156824Z","iopub.execute_input":"2021-09-04T09:07:49.157313Z","iopub.status.idle":"2021-09-04T09:07:49.167075Z","shell.execute_reply.started":"2021-09-04T09:07:49.157274Z","shell.execute_reply":"2021-09-04T09:07:49.165523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(patient_df['Age'])","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:50.077127Z","iopub.execute_input":"2021-09-04T09:07:50.077551Z","iopub.status.idle":"2021-09-04T09:07:50.64029Z","shell.execute_reply.started":"2021-09-04T09:07:50.077515Z","shell.execute_reply":"2021-09-04T09:07:50.639167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df.hist(by='Gender',column = 'Age')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:51.076189Z","iopub.execute_input":"2021-09-04T09:07:51.076549Z","iopub.status.idle":"2021-09-04T09:07:51.503367Z","shell.execute_reply.started":"2021-09-04T09:07:51.076518Z","shell.execute_reply":"2021-09-04T09:07:51.502365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df.hist(by='Gender',column = 'num_of_opacities')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:52.0864Z","iopub.execute_input":"2021-09-04T09:07:52.086795Z","iopub.status.idle":"2021-09-04T09:07:52.780178Z","shell.execute_reply.started":"2021-09-04T09:07:52.08676Z","shell.execute_reply":"2021-09-04T09:07:52.779126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_df.hist(by='Gender',column = 'class')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:53.156565Z","iopub.execute_input":"2021-09-04T09:07:53.156924Z","iopub.status.idle":"2021-09-04T09:07:53.514302Z","shell.execute_reply.started":"2021-09-04T09:07:53.156893Z","shell.execute_reply":"2021-09-04T09:07:53.513272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ncorr = patient_df.corr()\ncorr","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:54.342229Z","iopub.execute_input":"2021-09-04T09:07:54.343096Z","iopub.status.idle":"2021-09-04T09:07:54.361575Z","shell.execute_reply.started":"2021-09-04T09:07:54.343003Z","shell.execute_reply":"2021-09-04T09:07:54.359743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = patient_df.corr()\ncorr","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:55.331417Z","iopub.execute_input":"2021-09-04T09:07:55.331878Z","iopub.status.idle":"2021-09-04T09:07:55.347323Z","shell.execute_reply.started":"2021-09-04T09:07:55.331841Z","shell.execute_reply":"2021-09-04T09:07:55.346134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n\n1. There are more people not diagnosed with Pneumonia. Target attribute is distributed as\n\n     a. 0 - 20672 - 77.47\n\n     b. 1 - 6012 - 22.53\n\n2. Gender - Target -> Number of male patients is quite similar in count as the number of female patients for both Normal and Pneumonia affected, with male patients slightly higher.\n3. ViewPosition - Target -> Number of patients who have given CXR in PA viewposition is quite same in count as the number of patients in AP viewposition for both Normal and Pneumonia affected.\n4. Class - Target -> No Lung Opacity / Not Normal class is also considered as Non-Pneumonic along with Normal class. Lung Opacity confirms Positive Pneumonia in patients.","metadata":{}},{"cell_type":"markdown","source":"**Observations from the visualizations**","metadata":{}},{"cell_type":"code","source":"printmd('**We can see the data is distributed across the different classes with adequate representation from each class. \\\nThereby, we can use the existing dataset for model generation.**', color='blue')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:07:58.089622Z","iopub.execute_input":"2021-09-04T09:07:58.090032Z","iopub.status.idle":"2021-09-04T09:07:58.096874Z","shell.execute_reply.started":"2021-09-04T09:07:58.089973Z","shell.execute_reply":"2021-09-04T09:07:58.095869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**Model Building**","metadata":{}},{"cell_type":"markdown","source":"1) Building a pneumonia detection model starting from basic CNN and then improving upon it.","metadata":{}},{"cell_type":"code","source":"print('As we can have multiple bounding boxes for the lung opacity condition, we can consider a mask with all region with lung \\\nmarked by all 1s to indicate the presence of lung opacity. This can help us represent multiple lung opacity regions in a single mask.')\nprint('\\nAdditionally, we would be using a convolutional neural network based on resnet to receive high level of accuracy.')","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:08:24.465Z","iopub.execute_input":"2021-09-04T09:08:24.465505Z","iopub.status.idle":"2021-09-04T09:08:24.471625Z","shell.execute_reply.started":"2021-09-04T09:08:24.465458Z","shell.execute_reply":"2021-09-04T09:08:24.470399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print('As the data available for training is large, we can build a data generator class that can help in building the input data \\\nin batches to avoid overwhelming the system memory.')\nprint('We can extend the keras sequence class to get the data generator.')\n\nclass LungDataGenerator(keras.utils.Sequence):\n    def __init__(self, patientIds, batch_size=32, dim=(1024,1024), shuffle=True):\n        self.patientIds = patientIds\n        self.batch_size = batch_size\n        self.dim=dim\n        self.shuffle = shuffle\n        self.on_epoch_end()\n        \n    def __load__(self, patientId):\n        # Load the dicom file as a numpy array\n        dicom_image = pydicom.dcmread(trainDicomFiles[patientId]).pixel_array\n\n        # Create an empty mask to hold the list of bounding boxes as image segments.\n        bounding_box_mask = np.zeros(dicom_image.shape) \n\n        # Generate the mask using the label data frame\n        targetLabelDf = trainLabelAndBoundingBoxDf.loc[patientId]\n        if targetLabelDf['target'] == 1:\n            for box in targetLabelDf['boundingBox']:\n                x, y, width, height = [int(item) for item in box]\n                bounding_box_mask[y:y+height, x:x+width] = 1\n\n        # Resizing the image for reduction (to help build faster model).\n        dicom_image = cv2.resize(dicom_image, (self.dim[0], self.dim[1]))\n        bounding_box_mask = cv2.resize(bounding_box_mask, (self.dim[0], self.dim[1]))\n        dicom_image = np.expand_dims(dicom_image, -1)\n        bounding_box_mask = np.expand_dims(bounding_box_mask, -1)\n        target = targetLabelDf['target']\n        return dicom_image, bounding_box_mask, target\n\n    def __len__(self):\n        # Represents the number of batches per epoch\n        return int(len(self.patientIds) / self.batch_size)\n    \n    def __getitem__(self, index):\n        # Generates the indexes associated with one batch\n        targetPatientIds = self.patientIds[index * self.batch_size : (index + 1) * self.batch_size]\n        \n        items = [self.__load__(id) for id in targetPatientIds]\n\n        # Zip the images and masks for the target patients.\n        imgs, masks, targets = zip(*items)\n\n        # Create numpy batch\n        imgs = np.array(imgs)\n        masks = np.array(masks)\n        targets = np.array(targets)\n        return imgs, masks, targets\n        \n    def on_epoch_end(self):\n        # Shuffles the data after every epoch\n        if self.shuffle:\n            np.random.shuffle(self.patientIds)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:09:20.287695Z","iopub.execute_input":"2021-09-04T09:09:20.288123Z","iopub.status.idle":"2021-09-04T09:09:20.305634Z","shell.execute_reply.started":"2021-09-04T09:09:20.288088Z","shell.execute_reply":"2021-09-04T09:09:20.304195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Now we can define the metrics for loss reduction as below:')\nprint('Define IOU or jaccard loss function')\ndef iou_loss(y_true, y_pred):\n    y_true = tf.reshape(y_true, [-1])\n    y_pred = tf.reshape(y_pred, [-1])\n    intersection = tf.reduce_sum(y_true * y_pred)\n    score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + 1.)\n    return 1 - score\n\nprint('Combine BCE and IOU loss')\ndef iou_bce_loss(y_true, y_pred):\n    return 0.5 * keras.losses.binary_crossentropy(y_true, y_pred) + 0.5 * iou_loss(y_true, y_pred)\n    #return iou_loss(y_true, y_pred)\n\nprint('Get mean IOU as a metric')\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis = [1, 2, 3])\n    union = tf.reduce_sum(y_true, axis = [1, 2, 3]) + tf.reduce_sum(y_pred, axis = [1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:09:21.312305Z","iopub.execute_input":"2021-09-04T09:09:21.312691Z","iopub.status.idle":"2021-09-04T09:09:21.323546Z","shell.execute_reply.started":"2021-09-04T09:09:21.312656Z","shell.execute_reply":"2021-09-04T09:09:21.322522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('We would create the convolutional neural network using resnet blocks.\\\nThe model consists of a downsampling block (consisting of Batch normalization, leaky ReLU, Convolutional 2D and max pool layer) \\\nand a resnet block (consisting of batch normalization, leaky ReLU and convolutional 2D layers). The downsampling and resnet \\\nblocks are repeated based on configuration parameters of n_blocks and depth. In the end we upsample the data points to the same \\\nsize as before in order to get the target mask.')\ndef create_downsample(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 1, padding='same', use_bias=False)(x)\n    x = keras.layers.MaxPool2D(2)(x)\n    return x\n\ndef create_resblock(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    return keras.layers.add([x, inputs])\n\ndef create_network(input_size, channels, n_blocks=2, depth=4):\n    # Input\n    inputs = keras.Input(shape=(input_size, input_size, 1))\n\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(inputs)\n    # Residual blocks\n    for d in range(depth):\n        channels = channels * 2\n        x = create_downsample(channels, x)\n        for b in range(n_blocks):\n            x = create_resblock(channels, x)\n        \n    # Output\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(1, 1, activation='sigmoid')(x)\n    outputs = keras.layers.UpSampling2D(2**depth)(x)\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:09:22.202766Z","iopub.execute_input":"2021-09-04T09:09:22.203134Z","iopub.status.idle":"2021-09-04T09:09:22.217699Z","shell.execute_reply.started":"2021-09-04T09:09:22.203102Z","shell.execute_reply":"2021-09-04T09:09:22.216545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Generating the network instance.')\nmodel = create_network(input_size=256, channels=32, n_blocks=2, depth=4)\nmodel.compile(optimizer='adam', loss=iou_bce_loss, metrics=['accuracy', mean_iou])\n\nepoch_count = 5\n\n# Define cosine annealing\ndef cosine_annealing(x):\n    lr = 0.001\n    epochs = epoch_count\n    return lr * (np.cos(np.pi * x / epochs) + 1.) / 2\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:09:23.822001Z","iopub.execute_input":"2021-09-04T09:09:23.822443Z","iopub.status.idle":"2021-09-04T09:09:24.701501Z","shell.execute_reply.started":"2021-09-04T09:09:23.822399Z","shell.execute_reply":"2021-09-04T09:09:24.700413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training the model.')\nlearning_rate = tf.keras.callbacks.LearningRateScheduler(cosine_annealing)\n\ntraining_data_percent = 0.7\ntraining_data_end_index = int(trainLabelAndBoundingBoxDf.shape[0] * training_data_percent)\ntraining_patientIds_box = trainLabelAndBoundingBoxDf['patientId'][0:training_data_end_index]\nvalidation_patientIds_box = trainLabelAndBoundingBoxDf['patientId'][training_data_end_index:]\n\nparams = {'dim': (256, 256), 'batch_size': 32, 'shuffle': True}\ntraining_data_generator_box = LungDataGenerator(training_patientIds_box, **params)\nvalidation_data_generator_box = LungDataGenerator(validation_patientIds_box, **params)\n\nhistory = model.fit(training_data_generator_box, validation_data=validation_data_generator_box, callbacks=[learning_rate], epochs=epoch_count, workers=4, use_multiprocessing=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-04T09:09:24.992876Z","iopub.execute_input":"2021-09-04T09:09:24.993251Z","iopub.status.idle":"2021-09-04T09:55:08.272188Z","shell.execute_reply.started":"2021-09-04T09:09:24.99322Z","shell.execute_reply":"2021-09-04T09:55:08.268883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}