{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# ------------------------------------Loading the csv data==================\n\nimport pandas as pd\nDetailed_Class_info = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\nTrain_Labels = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\n\n# viewing detailed Class Info\n\nDetailed_Class_info.head()\n# print(Detailed_Class_info.count))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print (Detailed_Class_info['class'].unique)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Three category of classes found - No Lung Opacity / Not Normal, Normal and Lung Opacity"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Viewing train labels\nTrain_Labels.head()\n# print(Train_Labels.count())\n\n# 0 - No Pneumonia detected\n# 1 - Pneumonia detected with the co-ordinates of bounding box","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Removing Duplicates from both sets and joining the dataframes on patient_ID for better data visualization\n\nDetailed_Class_info = Detailed_Class_info.drop_duplicates('patientId').reset_index(drop=True)\n\nTrain_Labels = Train_Labels.drop_duplicates('patientId').reset_index(drop=True)\n\nData=Train_Labels.merge(Detailed_Class_info, how='inner', on='patientId')\n\nData.head(15)\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From above set below can be concluded \n1) No Lung Opacity/Not Normal class belongs to 0- Target Variable (No Pneumonia)\n2) Lung Opacity class belong to target variable 1 (Pneumonia)\n3) Normal class also belongs to target variable 0 (No Pneumonia)\n\nlooking at the set above we can say if Lung Opacity is observed in patient he/she is likely to have pneumonia"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's try to visualize our target vairable counts\n\nimport seaborn as sns\nsns.countplot(x=\"class\",hue=\"class\",data=Detailed_Class_info)\n\n# Approx 6k patients with Pneumonia detected\n# 12k approx have no lung opacity but classified as no pneumonia\n# 8k-10k normal cases","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Checking if the class imbalance is observed\n\nsns.countplot(x=\"Target\",hue=\"Target\",data=Train_Labels)\n# Looking at the counts below data imbalance is found in 0 and 1 classes\n# We have more data with No Pneumonia detected cases","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Visualizing images from the dataset\nimport pydicom as dcm\nfrom pydicom import dcmread","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Copying all files with .dcm extension into a list of train images\n\nimport glob \nTrain_image_list= glob.glob('../input/rsna-pneumonia-detection-challenge/stage_2_train_images/*.dcm')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Reading all the image data from the list of dcm files\nTrain_Images = [dcm.read_file(x, stop_before_pixels=False) for x in Train_image_list]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Checking the data type of the image \n\n# Its a DICOM image with tags written to it along with image\ntype(Train_Images[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's try to print image data\nprint(Train_Images[1])\n\n# Some imoortant tags can be seen from the data like Patient_ID ,Patient_Name","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pylab\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Visualizing the images with patient id's in as X labels\n\nfig=pylab.figure()\n\nfig.set_size_inches(15,5)\n\nfig1=fig.add_subplot(1,4,1)\n\nfig1.set_xlabel(Train_Images[0].PatientID)\n\nfig2=fig.add_subplot(1,4,2)\n\nfig2.set_xlabel(Train_Images[1].PatientID)\n\nfig3=fig.add_subplot(1,4,3)\n\nfig3.set_xlabel(Train_Images[2].PatientID)\n\nfig4=fig.add_subplot(1,4,4)\n\nfig4.set_xlabel(Train_Images[3].PatientID)\n\nfig1.imshow(Train_Images[0].pixel_array, cmap=pylab.cm.bone)\n\nfig2.imshow(Train_Images[1].pixel_array, cmap=pylab.cm.bone)\n\nfig3.imshow(Train_Images[2].pixel_array, cmap=pylab.cm.bone)\n\nfig4.imshow(Train_Images[3].pixel_array, cmap=pylab.cm.bone)\n\nPatient1=Data[Data['patientId']==Train_Images[0].PatientID]\nPatient2=Data[Data['patientId']==Train_Images[1].PatientID]\nPatient3=Data[Data['patientId']==Train_Images[2].PatientID]\nPatient4=Data[Data['patientId']==Train_Images[3].PatientID]\n\nprint(\"Patient 1---->\",Patient1['Target'])\nprint(\"Patient 2---->\",Patient2['Target'])\nprint(\"Patient 3---->\",Patient3['Target'])\nprint(\"Patient 4---->\",Patient4['Target'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# From the images above its evident that if lung opacity is obhserved its like to have pneumonia.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Patient3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Visualizing Image with bounding box over the affected area as per the data\n\nimport matplotlib.patches as patches\n\nfig=pylab.figure()\n\nfig.set_size_inches(15,5)\n\nfig1=fig.add_subplot(1,1,1)\n\nfig1.imshow(Train_Images[2].pixel_array, cmap=pylab.cm.bone)\n\nrect = patches.Rectangle((321,246),285,525, edgecolor='r', facecolor=\"none\")\n\nfig1.add_patch(rect)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's capture lables and extract pixel data from images\n \n\nprint(Data.shape)\nprint(len(Train_Images))\n\n# Checking if the data in csv and images are in same sequence\n\nprint(Data.head(1))\nprint(Train_Images[0].PatientID)\n\n# Images are not in the same sequence","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}