{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Introduction to problem statement\nPneumonia is an infection in one or both lungs. Bacteria, viruses, and fungi cause it. \nThe infection causes inflammation in the air sacs in your lungs, which are called alveoli.\n\nNow to detection Pneumonia we need to detect Inflammation of the lungs. \nIn this project, you’re challenged to build an algorithm to detect a visual signal for pneumonia in medical images. \nSpecifically, your algorithm needs to automatically locate lung opacities on chest radiographs.\n\nBusiness Domain Value\nAutomating Pneumonia screening in chest radiographs, providing affected area details through bounding box. \nAssist physicians to make better clinical decisions or even replace human judgement in certain functional areas of healthcare (eg, radiology).\n\nProject objective\nIn this capstone project, the goal is to build a pneumonia detection system, to locate the\nposition of inflammation in an image.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Importing all the standard libraries\n#..... array/martrix operations and dataframe libraries\nimport numpy as np\nimport pandas as pd\n#...........\n#.......... Visulaization libraries\nimport pydicom\nimport pylab\nimport matplotlib.pyplot as plt\nimport seaborn as sn\nfrom skimage.transform import resize\n\n#......\nfrom sklearn.model_selection import train_test_split\n\n# NN model building linraries\nfrom keras.utils import to_categorical\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D\nfrom keras.layers import MaxPooling2D\nfrom keras.layers import Flatten\nfrom keras.layers import Dense\nfrom keras.layers import Dropout\n#...................................................","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# setting path for each of the files\nclass_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv'\nlabels_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\nImage_train_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Files descrition\n#1. stage_2_detailed_class_info.csv- contains the information of target label\n#2. stage_2_train_labels.csv- contains information on Target and bounding box\n#3. stage_2_train_images- contains training images in dcm format","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploration of Given Data,classes and images of different classes*****"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Reading class file (first file) as dataframe and check few entries and shape\ndf_class=pd.read_csv(class_path)\nprint(df_class.head(10))\nprint(df_class.shape[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_class['class'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Observation:\n# This file ocntains patient Id and repective class ifnormation. \n#. There are 30277 records\n# There are three classes- \n#    1. Lung Opacity- Patient havinig pneumonia, \n#    2. Normal- Patient not having pnemonia and not having any other lung problem\n#    3. No Lung Opacity/Not Normal- Patient not having pnemonia but having any other lung problem","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_class.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Observation- There are no null values ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# checking the number of unique entries with respect to patient ID\nprint(df_class['patientId'].value_counts().shape[0],'patient cases')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Reading label file (second file) as dataframe and check few entries and shape\ndf_label=pd.read_csv(labels_path)\nprint(df_label.head())\nprint(df_label.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Observation\n#1. There are 30277 lables record (same as the class dataframe)\n#2. There are 6 columns - pateint ID (same as order as in class dataframe), bounding box co-ordinates, height and widht and Target label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df_label.info())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Obervations- There are null values (NaN) in x,y, widht and height","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Now lets drop the duplicate cases\ndf=pd.concat([df_Label,df_class.drop('patientId',1)],1)\nprint(df.shape)\nprint(df.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Classes and Targets based on Patient count\ndf.groupby(['class','Target']).size().reset_index(name='patient_numbers')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number of duplicate entries accross rows:\\n', df[df.duplicated()].count())\nprint('Number of duplicate Patient Id entries :\\n', df[df.duplicated(subset='patientId')].count())\nprint('Number of unique Patient Id entries: \\n', df['patientId'].nunique())\nprint('Count of various classes: \\n',df.groupby('class')['patientId'].nunique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Observation\n#1. All the Normal and No Lung Opacity / Not Normal\tpatients are grouped under Target label 0 (no pnemonia)\n#2. Data Imabalance- there are ~30% pneumonia records and rest ~70% no pneumonia\n#3  There are no duplicates accross rows\n#4. Checking for duplicate patientId's, there are 26684 unique Patient Ids","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#--------------------------------------- Exploring training images data -------------------","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# chekcing the type of image file format and total number of images\nimage_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'\nprint(os.listdir(image_path)[0])\nimport glob\nprint(len(list(glob.iglob(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/*.dcm\", recursive=True))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Observations:\n# All the images are in dcm format \n# these image file saved in the Digital Imaging and Communications in Medicine (DICOM) image format. \n#It stores a medical image, such as a CT scan or ultrasound\n# There are in total 26684 images which matches with the unique patient IDs. Seems there is no missing image file","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Checking sample image file for first entry in dataframe\nprint(df.iloc[0])\npatientId = df['patientId'][0]\nimage_path_1='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' %patientId\ndcm_data=pydicom.read_file(image_path_1)\nprint(dcm_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Observations:\n# dcm file contains metadata information about Patient (sample with no pnemonia): \n#             name, ID, Age, Sex, body part examines, view position, pixel data of image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#size of image\ndcm_data.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Plotting the image \nplt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('Pateint- No pneumonia class')\nplt.imshow(dcm_data.pixel_array)\nplt.subplot(122)\nplt.title('Pateint- No pneumonia class')\nplt.imshow(dcm_data.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets us plot one Patient with pnemonia (Target = 1)\nprint(df.iloc[4])\npatientId = df['patientId'][4]\nimage_path_1='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' %patientId\ndcm_data=pydicom.read_file(image_path_1)\nprint(dcm_data)\n#Plotting the image \nplt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('Pateint- With pneumonia class')\nplt.imshow(dcm_data.pixel_array)\nplt.subplot(122)\nplt.title('Pateint- With pneumonia class')\nplt.imshow(dcm_data.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patientId = df['patientId'][0]\ndcm_file = '/content/drive/My Drive/Colab Notebooks/Capstone/stage_2_train_images/%s.dcm' % patientId\ndcm_data=pydicom.read_file(dcm_file)\nprint(dcm_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/0004cfab-14fd-4e49-80ba-63a80b6bddd6.dcm'\nImage_1=pydicom.read_file(image_path)\nImage_1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Image_1.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets check the view of the lung\nplt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('color scale image')\nplt.imshow(Image_1.pixel_array)\nplt.subplot(122)\nplt.title('gray scale image')\nplt.imshow(Image_1.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Patient who is normal and image\nDF_class.iloc[3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path_1='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/0004cfab-14fd-4e49-80ba-63a80b6bddd6.dcm'\ndcm_1_data=pydicom.read_file(image_path_1)\ndcm_1_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_1_data.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets check the view of the lung\nplt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('color scale image')\nplt.imshow(dcm_1_data.pixel_array)\nplt.subplot(122)\nplt.title('gray scale image')\nplt.imshow(dcm_1_data.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Now finally lets check the patient who is having pneumonia\nDF_class.iloc[4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path_2='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/00436515-870c-4b36-a041-de91049b9ab4.dcm'\ndcm_2_data=pydicom.read_file(image_path_2)\ndcm_2_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_2_data.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('color scale image')\nplt.imshow(dcm_2_data.pixel_array)\nplt.subplot(122)\nplt.title('gray scale image')\nplt.imshow(dcm_2_data.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets check how many images are there\nimage_data=os.listdir(Image_train_path)\nlen(image_data)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#getting useful information from images\npatient_data=[]\nfor i in DF_Label['patientId']:\n    patient_data_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % i\n    patient_image_data=pydicom.read_file(patient_data_path)\n    patient_data.append([i,\n                         patient_image_data.PatientAge,\n                         patient_image_data.PatientSex,\n                         patient_image_data.ViewPosition,\n                         patient_image_data.Rows,\n                         patient_image_data.Columns])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_data[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_patient_data=pd.DataFrame(data=patient_data,columns=['patientId','patientAge','patientSex','patient_View_position',\n                                                        'pixel_rows','pixel_columns'])\nDF_patient_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_patient_data['patientAge']=DF_patient_data['patientAge'].apply(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_patient_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Now lets combine all the dataset\nDF_Full=pd.concat([DF_patient_data,DataFrame_Comb],axis=1)\nDF_Full.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets drop the duplicate columns\nDF_Full=DF_Full.loc[:,~DF_Full.columns.duplicated()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check any missing values"},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Visualisation of different classes"},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full['class'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full['Target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full['patientSex'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Building the model\nDF_Full.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"resized_shape=(64,64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Creating pixel columns\npixel_labels=[]\nfor i in range(resized_shape[0]*resized_shape[1]):\n    pixel_labels.append(\"pixel\"+str(i))\npixel_labels[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total_images=DF_Full.shape[0]\n# Creating 1D array for all images\npixel_data=[]\nnum=0\nfor i in range(DF_Full.shape[0]):\n    patientId=DF_Full.iloc[i]['patientId']\n    dcm_file= '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % patientId\n    dcm_data=pydicom.read_file(dcm_file)\n    image=dcm_data.pixel_array\n    \n    final_pixel_array=[]\n    for j in resize(image,resized_shape):\n        final_pixel_array.extend(j)\n    pixel_data.append(final_pixel_array)\n    num=num+1\n    if num==total_images:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=pd.DataFrame(data=pixel_data,columns=pixel_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y=DF_Full['Target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X.shape,y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}