{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"pwd\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ls /kaggle/working","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv'\nlabels_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\nImage_train_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'\nImage_test_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images/'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploration of Given Data,classes and images of different classes*****","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_class=pd.read_csv(class_path)\nDF_class.head(10)\nprint(DF_class.shape[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(DF_class['patientId'].value_counts().shape[0],'patient cases')\nDF_class.groupby('class').size()\nDF_class.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Analysis of stage 2 patients labels information\nDF_Label=pd.read_csv(labels_path)\nprint(DF_Label.shape[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(DF_Label['patientId'].value_counts().shape[0])\nDF_Label.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Lets merge two data's\nDataFrame=pd.merge(DF_class,DF_Label,on='patientId')\nprint(DataFrame.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Now lets drop the duplicate cases\nDataFrame_Comb=pd.concat([DF_Label,DF_class.drop('patientId',1)],1)\nprint(DataFrame_Comb.shape[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DataFrame_Comb.sample(10)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DataFrame_Comb.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Classes and Targets based on Patient count\nDataFrame_Comb.groupby(['class','Target']).size().reset_index(name='patient_numbers')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\nimport pylab\nimport matplotlib.pyplot as plt\nimport seaborn as sn","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Now lets read the image\n# First lets check the image of a person with no lung opacity but not normal\nDF_class.iloc[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/0004cfab-14fd-4e49-80ba-63a80b6bddd6.dcm'\nImage_1=pydicom.read_file(image_path)\nImage_1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Image_1.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets check the view of the lung\nplt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('color scale image')\nplt.imshow(Image_1.pixel_array)\nplt.subplot(122)\nplt.title('gray scale image')\nplt.imshow(Image_1.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Patient who is normal and image\nDF_class.iloc[3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path_1='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/0004cfab-14fd-4e49-80ba-63a80b6bddd6.dcm'\ndcm_1_data=pydicom.read_file(image_path_1)\ndcm_1_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_1_data.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# lets check the view of the lung\nplt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('color scale image')\nplt.imshow(dcm_1_data.pixel_array)\nplt.subplot(122)\nplt.title('gray scale image')\nplt.imshow(dcm_1_data.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Now finally lets check the patient who is having pneumonia\nDF_class.iloc[4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path_2='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/00436515-870c-4b36-a041-de91049b9ab4.dcm'\ndcm_2_data=pydicom.read_file(image_path_2)\ndcm_2_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dcm_2_data.pixel_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(12,10))\nplt.subplot(121)\nplt.title('color scale image')\nplt.imshow(dcm_2_data.pixel_array)\nplt.subplot(122)\nplt.title('gray scale image')\nplt.imshow(dcm_2_data.pixel_array,cmap=plt.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets check how many images are there\nimage_data=os.listdir(Image_train_path)\nlen(image_data)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#getting useful information from images\npatient_data=[]\nfor i in DF_Label['patientId']:\n    patient_data_path='/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % i\n    patient_image_data=pydicom.read_file(patient_data_path)\n    patient_data.append([i,\n                         patient_image_data.PatientAge,\n                         patient_image_data.PatientSex,\n                         patient_image_data.ViewPosition,\n                         patient_image_data.Rows,\n                         patient_image_data.Columns])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_data[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_patient_data=pd.DataFrame(data=patient_data,columns=['patientId','patientAge','patientSex','patient_View_position',\n                                                        'pixel_rows','pixel_columns'])\nDF_patient_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_patient_data['patientAge']=DF_patient_data['patientAge'].apply(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_patient_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Now lets combine all the dataset\nDF_Full=pd.concat([DF_patient_data,DataFrame_Comb],axis=1)\nDF_Full.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets drop the duplicate columns\nDF_Full=DF_Full.loc[:,~DF_Full.columns.duplicated()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check any missing values","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Visualisation of different classes","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full['class'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full['Target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DF_Full['patientSex'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from skimage.transform import resize\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D\nfrom keras.layers import MaxPooling2D\nfrom keras.layers import Flatten\nfrom keras.layers import Dense\nfrom keras.layers import Dropout","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Building the model\nDF_Full.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"resized_shape=(64,64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patientId=DF_Full['patientId'][1]\nprint(DF_Full['patientId'][1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Creating pixel columns\npixel_labels=[]\nfor i in range(resized_shape[0]*resized_shape[1]):\n    pixel_labels.append(\"pixel\"+str(i))\npixel_labels[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(resized_shape[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total_images=DF_Full.shape[0]\n# Creating 1D array for all images\npixel_data=[]\nnum=0\nfor i in range(DF_Full.shape[0]):\n    patientId=DF_Full.iloc[i]['patientId']\n    dcm_file= '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % patientId\n    dcm_data=pydicom.read_file(dcm_file)\n    image=dcm_data.pixel_array\n    \n    final_pixel_array=[]\n    for j in resize(image,resized_shape):\n        final_pixel_array.extend(j)\n    pixel_data.append(final_pixel_array)\n    num=num+1\n    if num==total_images:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=pd.DataFrame(data=pixel_data,columns=pixel_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y=DF_Full['Target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X.shape,y.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(X,y,stratify=y,random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train_c = to_categorical(y_train)\ny_test_c = to_categorical(y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train_re = X_train.values.reshape(X_train.shape[0], resized_shape[0], resized_shape[1], 1)\nX_test_re = X_test.values.reshape(X_test.shape[0], resized_shape[0], resized_shape[1], 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\n\n# First Convolution  Layer \nmodel.add(Conv2D(filters = 6,\n                 kernel_size = 3,\n                 activation = 'relu',\n                 input_shape = (resized_shape[0], resized_shape[1], 1)\n                ))\n\n# Adding pooling\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\n# Second convolutional layer\nmodel.add(Conv2D(filters=16,\n                 kernel_size=3,\n                 activation='relu'))\n\n# Adding Pooling\nmodel.add(MaxPooling2D(pool_size=(2,2)))\n\n# Dropout\nmodel.add(Dropout(0.5))\n\n# Flatten\nmodel.add(Flatten())\n\n#Third Convoltion Layer\nmodel.add(Dense(512,\n                activation='relu'))\n\n# Fourth Convoltion layer\nmodel.add(Dense(128,\n                activation='relu'))\n# Dropout\nmodel.add(Dropout(0.5))\n\n#Output Layer\nmodel.add(Dense(2, activation='softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss = 'categorical_crossentropy', \n              optimizer = 'adam', \n              metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trained_model = model.fit(X_train_re,\n                          y_train_c,\n                          batch_size = 32,\n                          validation_data = (X_test_re, y_test_c),\n                          epochs = 20,\n                          verbose = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}