{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Please do Vote up if you liked my work\n\nMy [Linkedin](https://www.linkedin.com/in/letian-dai-phd-physics-nanomaterial-nanoscience-nanotechnology-datascience-bigdata/) <br>\nMy [Git](https://github.com/daiwofei)","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n#visualisation\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nimport glob\nimport scipy.ndimage as ndimage\nfrom skimage import measure, morphology, segmentation\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\n\n#read the .dcm file\nimport pydicom\n\nfrom scipy.stats import probplot, mode\n\n#color\nfrom colorama import Fore, Back, Style\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1. Input the data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check the list of files or folders in the data source\nlist(os.listdir(\"../input/osic-pulmonary-fibrosis-progression\"))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# input the data\ntrain_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\n\nprint(Fore.RED + 'Training data shape: ',Style.RESET_ALL,train_df.shape)\nprint(Fore.BLUE + 'Test data shape: ',Style.RESET_ALL,test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# preview of the train dataframe\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# preview of the test dataframe\ntest_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Show the list of columns\ncolumns = train_df.keys()\ncolumns = list(columns)\nprint(Fore.BLUE + \"List of columns in the train_df\",Fore.RED + \"\", columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if there is missing data in the dataframe\n# check the null part in the whole data set, red part is missing data, blue is non-null\nsns.heatmap(train_df.isnull(),yticklabels=False,cbar=False,cmap='coolwarm')\ntrain_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There is no missing value in train_df and test_df.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the null part in the whole data set, red part is missing data, blue is non-null\nsns.heatmap(test_df.isnull(),yticklabels=False,cbar=False,cmap='coolwarm')\ntest_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the type of dataframe\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. Exploratary Data Analysis  ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check the unique number of patients' ids in train dataframe\n\nprint(Fore.WHITE + \"In the train_df,\",Fore.RED + \"the total patient ids are\",Style.RESET_ALL,f\"{train_df['Patient'].count()},\"\n      , Fore.BLUE + \"from those the unique ids are\", Style.RESET_ALL, f\"{train_df['Patient'].value_counts().shape[0]}.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check the unique number of patients' ids in test dataframe\n\nprint(Fore.WHITE + \"In the test_df,\",Fore.RED + \"the total patient ids are\",Style.RESET_ALL,f\"{test_df['Patient'].count()},\"\n      , Fore.BLUE + \"from those the unique ids are\", Style.RESET_ALL, f\"{test_df['Patient'].value_counts().shape[0]}.\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"All patients are unique in the test_df. ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# compare the test patients' ids and train patients' ids\n\ntrain_patient_ids = set(train_df['Patient'].unique())\ntest_patient_ids = set(test_df['Patient'].unique())\n\n# get the intersection of test and train datasets\ntest_patient_ids.intersection(train_patient_ids)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The whole 5 patients in test_df are found in train_df as well.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# The histogram of patients' samples (how many samples are in the same patient) \n\nplt.figure(figsize=(6,4))\ntrain_df['Patient'].value_counts().hist(alpha=0.5,color='green',label='Samples for the same patient')\nplt.legend()\n\nplt.xlabel('Samples for the same patient')\nplt.ylabel('Patients')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## It is necessary to organize the data of the same patient and create individual patient dataframe","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's verify the features for the same patient, for example the patient with id = ID00419637202311204720264\n\ntrain_df[train_df['Patient'] == 'ID00419637202311204720264']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's verify the features for the same patient, for example the patient with id = ID00421637202311550012437\n\ntrain_df[train_df['Patient'] == 'ID00421637202311550012437']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's verify the features for the same patient, for example the patient with id = ID00422637202311677017371\n\ntrain_df[train_df['Patient'] == 'ID00422637202311677017371']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's verify the features for the same patient, for example the patient with id = ID00423637202312137826377\n\ntrain_df[train_df['Patient'] == 'ID00423637202312137826377']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Let's verify the features for the same patient, for example the patient with id = ID00426637202313170790466\n\ntrain_df[train_df['Patient'] == 'ID00426637202313170790466']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# From the description of data: \"In the dataset, you are provided with a baseline chest CT scan and associated clinical information for a set of patients. A patient has an image acquired at time Week = 0 and has numerous follow up visits over the course of approximately 1-2 years, at which time their FVC is measured.\" *It is possible to get the initial FVC for each patient at the Week = 0.*","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# verify the relation between the 'FVC' and the 'Percent' for the same patient, for example the first patient with id = ID00123637202217151272140\n\nsns.regplot(x=\"FVC\", y=\"Percent\", data=train_df[train_df['Patient'] == 'ID00123637202217151272140'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the initial FVC at the week = 0, for example the first patient with id = ID00123637202217151272140\n# FVC is integer, so it is necessary to use function .round()\n\ninitial_FVC = ((train_df[train_df['Patient'] == 'ID00123637202217151272140']['FVC'].iloc[0]) / (train_df[train_df['Patient']== 'ID00123637202217151272140']['Percent'].iloc[0])*100).round() \n\nprint('The FVC of the patient with id = ID00123637202217151272140 is',Fore.BLUE + '', initial_FVC)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the initial FVC at the week = 0, for example the first patient with id = ID00009637202177434476278\n# FVC is integer, so it is necessary to use function .round()\n\ninitial_FVC = ((train_df[train_df['Patient'] == 'ID00009637202177434476278']['FVC'].iloc[0]) / (train_df[train_df['Patient']== 'ID00009637202177434476278']['Percent'].iloc[0])*100).round() \n\nprint('The FVC of the patient with id = ID00009637202177434476278 is',Fore.BLUE + '', initial_FVC)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## It is clear to see that the features of \"Age\", \"Sex\" and \"SmokingStatus\" is constant for the same patient.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create individual patient dataframe\npatient_df = train_df[['Patient', 'Age', 'Sex', 'SmokingStatus']].drop_duplicates()\npatient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# create a new row for each patient about the initial FVC at the week = 0\n# # iterating the columns \ni = 0\nInit_FVC = []\nfor row in patient_df.index: \n    ID = patient_df['Patient'].loc[row]\n    temp_FVC = ((train_df[train_df['Patient'] == ID]['FVC'].iloc[0]) / (train_df[train_df['Patient']== ID]['Percent'].iloc[0])*100).round() \n    Init_FVC.append(temp_FVC)\n    print(i,ID) \n    i = i+1\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# add the initial FVC inside the patient_df \npatient_df['FVC'] = Init_FVC\npatient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The corresponding Weeks is 0\npatient_df['Weeks'] = 0\npatient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The corresponding Percent is 100%\npatient_df['Percent'] = 100\npatient_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the sex elemnts in histogram\n# The Histogram of sex\npatient_df['Sex'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'Sex',linecolor='black',opacity=0.7,color='green',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the Sex column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the SmokingStatus elemnts in histogram\n# The Histogram of SmokingStatus\npatient_df['SmokingStatus'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'SmokingStatus',linecolor='black',opacity=0.7,color='red',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the SmokingStatus column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the age distribution in histogram\n# The Histogram of age\nplt.figure(figsize=(6,4))\npatient_df['Age'].hist(alpha=0.5,color='blue',label='Age', bins = 30)\nplt.legend()\n\nplt.xlabel('Age')\nplt.ylabel('Count')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the FVC distribution in histogram\n# The Histogram of FVC\nplt.figure(figsize=(6,4))\npatient_df['FVC'].hist(alpha=0.5,color='brown',label='FVC(Week=0)', bins = 30)\nplt.legend()\n\nplt.xlabel('FVC(Week=0)')\nplt.ylabel('Count')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# I need to convert the string to integer of \"Sex\" and \"SmokingStatus\"","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# convert the \"Ex-smoker\" = 1, \"Never smoked\" = 0, \"Currently smokes\" = 2 in the \"SmokingStatus\"\nSmoking_list = []\nfor i in np.arange(len(patient_df)):\n    status = patient_df['SmokingStatus'].iloc[i]\n    if status == 'Ex-smoker':\n        Smoking_list.append(1)\n    elif status == 'Never smoked':\n        Smoking_list.append(0)\n    else:\n        Smoking_list.append(2)\n\npatient_df['SmokingStatus'] = Smoking_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# convert the \"Male\" = 1, \"Female\" = 0 in the \"Sex\"\nSex_list = []\nfor i in np.arange(len(patient_df)):\n    gender = patient_df['Sex'].iloc[i]\n    if gender == 'Male':\n        Sex_list.append(1)\n    else:\n        Sex_list.append(0)\n\npatient_df['Sex'] = Sex_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## Because we have obtained the inital FVC at the week = 0, when the patient took the CT scan of their chest. So I need to get the relation betweeen the CT image and the corresponding FVC.  ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# The test_df gives the initial state of each patient with the scan CT images in the test folder. There is a big question needs to be answered \"Can we deduce the FVC via analyzing the CT images?\" \n\nIn the train and test folders, there are multiple `.dcm` files corresponding to different slices of CT scan for each patient measured at Week = 0. CT scans produce 3D volumes consist of 2D slices for the chest which should have direct relation between the FVC at the week = 0. Each scan is a 2D slice which is a `.dcm` file. ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#check the information of file '1.dcm' for the patient with id = ID00228637202259965313869\nfile_path = '../input/osic-pulmonary-fibrosis-progression/train/ID00228637202259965313869/1.dcm'\ndicom_file = pydicom.dcmread(file_path)\n\nprint(f'Patient: ID00228637202259965313869 Image: 1.dcm Dataset\\n{\".\" * 56}\\n\\n{dicom_file}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#check the information of file '1.dcm' for the patient with id = ID00228637202259965313869\nfile_path = '../input/osic-pulmonary-fibrosis-progression/train/ID00228637202259965313869/2.dcm'\ndicom_file = pydicom.dcmread(file_path)\n\nprint(f'Patient: ID00228637202259965313869 Image: 2.dcm Dataset\\n{\".\" * 56}\\n\\n{dicom_file}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#check the information of file '1.dcm' for the patient with id = ID00062637202188654068490\nfile_path = '../input/osic-pulmonary-fibrosis-progression/train/ID00062637202188654068490/1.dcm'\ndicom_file = pydicom.dcmread(file_path)\n\nprint(f'Patient: ID00422637202311677017371 Image: 1.dcm Dataset\\n{\".\" * 56}\\n\\n{dicom_file}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#check the information of file '1.dcm' for the patient with id = ID00009637202177434476278\nfile_path = '../input/osic-pulmonary-fibrosis-progression/train/ID00009637202177434476278/1.dcm'\ndicom_file = pydicom.dcmread(file_path)\ndicom_file.dir()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the relation between the 'SliceThickness', 'SingleCollimationWidth' and the difference of 'ImagePosition Patient of Z' or 'SliceLocation'\npatient_name = 'ID00228637202259965313869'\npatient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\nprint (len(patient_directory))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the relation between the 'SliceThickness', 'SingleCollimationWidth' and the difference of 'ImagePosition Patient of Z' or 'SliceLocation'\npatient_name = 'ID00228637202259965313869'\npatient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\nfor name in patient_directory:\n    eachslice = pydicom.dcmread(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}/{name}')\n    print (eachslice.ImagePositionPatient[2], '  ',eachslice.SliceLocation,'   ', eachslice.SliceThickness, '   ', eachslice.SingleCollimationWidth,'    ', eachslice.TableSpeed)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the relation between the 'SliceThickness', 'SingleCollimationWidth' and the difference of 'ImagePosition Patient of Z' or 'SliceLocation'\npatient_name = 'ID00422637202311677017371'\npatient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\nfor name in patient_directory:\n    eachslice = pydicom.dcmread(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}/{name}')\n    print (eachslice.ImagePositionPatient[2], '  ',eachslice.SliceLocation,'   ', eachslice.SliceThickness, '   ', eachslice.SingleCollimationWidth,'    ', eachslice.TableSpeed)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It is clear to see in the case of patient with id = 'ID00422637202311677017371'. <br>\nThe 'ImagePositionPatient[2]' is the same as the 'SliceLocation'. <br>\nThe 'SliceThickness' is the same as the 'SingleCollimationWidth'.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_name = 'ID00007637202177411956430'\npatient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\nfor name in patient_directory:\n    eachslice = pydicom.dcmread(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}/{name}')\n    print (eachslice.ImagePositionPatient[2], '  ',eachslice.SliceLocation,'   ', eachslice.SliceThickness, '   ')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# the first patient in the test file\npatient_name = 'ID00419637202311204720264'\npatient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/test/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\nprint (len(patient_directory))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/schlerp/getting-to-know-dicom-and-the-data\ndef show_dcm_info(dataset):\n    print(Fore.YELLOW + \"Filename.........:\",Style.RESET_ALL,file_path)\n    print()\n\n    pat_name = dataset.PatientName\n    display_name = pat_name.family_name + \", \" + pat_name.given_name\n    print(Fore.BLUE + \"Patient's name......:\",Style.RESET_ALL, display_name)\n    print(Fore.BLUE + \"Patient id..........:\",Style.RESET_ALL, dataset.PatientID)\n    print(Fore.BLUE + \"Patient's Sex.......:\",Style.RESET_ALL, dataset.PatientSex)\n    print(Fore.YELLOW + \"Modality............:\",Style.RESET_ALL, dataset.Modality)\n    print(Fore.GREEN + \"Body Part Examined..:\",Style.RESET_ALL, dataset.BodyPartExamined)\n    \n    if 'PixelData' in dataset:\n        rows = int(dataset.Rows)\n        cols = int(dataset.Columns)\n        print(Fore.BLUE + \"Image size.......:\",Style.RESET_ALL,\" {rows:d} x {cols:d}, {size:d} bytes\".format(\n            rows=rows, cols=cols, size=len(dataset.PixelData)))\n        if 'PixelSpacing' in dataset:\n            print(Fore.YELLOW + \"Pixel spacing....:\",Style.RESET_ALL,dataset.PixelSpacing)\n            dataset.PixelSpacing = [1, 1]\n        plt.figure(figsize=(10, 10))\n        plt.imshow(dataset.pixel_array, cmap='gray')\n        plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_name = 'ID00419637202311204720264'\ntrain_file_path = (f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}/10.dcm')\ntest_file_path = (f'../input/osic-pulmonary-fibrosis-progression/test/{patient_name}/10.dcm')\n\ntrain_dataset = pydicom.dcmread(train_file_path)\ntest_dataset = pydicom.dcmread(test_file_path)\n\n\n\nf, (ax1, ax2) = plt.subplots(1, 2, sharey=True, figsize = (12, 12))\n\nax1.imshow(train_dataset.pixel_array, cmap='gray')\nax1.set_title(\"Train file\")\nax1.axis('off')\n\nax2.imshow(test_dataset.pixel_array, cmap='gray')\nax2.set_title(\"Test file\")\nax2.axis('off')\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# The CT files are the same for the same patient  in both train and test folders.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_name = 'ID00007637202177411956430'\npatient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\nfor name in patient_directory:\n    eachslice = pydicom.dcmread(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}/{name}')\n    print (eachslice.ImagePositionPatient[2], '  ', eachslice.SliceThickness)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_df2 = patient_df.copy()\n\ndef get_metadata(patient_name):\n    \n    patient_directory = sorted(os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}')\n                               , key=(lambda f: int(f.split('.')[0])))\n    first_slice = pydicom.dcmread(f'../input/osic-pulmonary-fibrosis-progression/train/{patient_name}/{patient_directory[0]}')\n\n    \n    features = ['Manufacturer','TotalCollimationWidth','SingleCollimationWidth','TableSpeed','KVP','Columns','Rows','DistanceSourceToDetector',\n               'DistanceSourceToPatient','GeneratorPower','HighBit','PixelRepresentation','SliceLocation','SliceThickness','TableHeight',\n               'RevolutionTime','PatientPosition','XRayTubeCurrent']\n    for feature in features:\n        if feature in first_slice.dir():\n             patient_df2.loc[patient_df2['Patient'] == patient_name, feature] = first_slice.get(feature)\n    \n    \n    patient_df2.loc[patient_df2['Patient'] == patient_name, 'PixelSpacing'] = first_slice.PixelSpacing[0]\n    \n    \n    if 'ImagePositionPatient' in first_slice.dir():\n        patient_df2.loc[patient_df2['Patient'] == patient_name, 'ImagePositionPatient_X'] = first_slice.ImagePositionPatient[0]\n        patient_df2.loc[patient_df2['Patient'] == patient_name, 'ImagePositionPatient_Y'] = first_slice.ImagePositionPatient[1]\n        patient_df2.loc[patient_df2['Patient'] == patient_name, 'ImagePositionPatient_Z'] = first_slice.ImagePositionPatient[2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for patient in patient_df2['Patient']:\n    get_metadata(patient)\n\npatient_df2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the Manufacturer distribution in histogram\nfig = px.histogram(patient_df2, x=\"Manufacturer\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the SingleCollimationWidth distribution in histogram\nfig = px.scatter(patient_df2, x=\"SingleCollimationWidth\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the TotalCollimationWidth distribution in histogram\nfig = px.scatter(patient_df2, x=\"TotalCollimationWidth\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the SliceThickness distribution in histogram\nfig = px.scatter(patient_df2, x=\"SliceThickness\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the RevolutionTime distribution in histogram\nfig = px.histogram(patient_df2, x=\"RevolutionTime\")\nfig.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the TableSpeed distribution in scatter\nfig = px.scatter(patient_df2, x=\"TableSpeed\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the PatientPosition distribution in scatter\nfig = px.scatter(patient_df2, x=\"PatientPosition\")\nfig.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the SliceLocation distribution in scatter\nfig = px.scatter(patient_df2, x=\"SliceLocation\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"# It is necessary to get the area of the lung in each slice and intergrate to obtain the volume of the whole chest. \n\nOne method of obtaining the area of the lung in each slice is to use the Marker-Controlled Watershed approach. https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed\ndef load_scan(path):\n    \"\"\"\n    Loads scans from a folder and into a list.\n    \n    Parameters: path (Folder path)\n    \n    Returns: slices (List of slices)\n    \"\"\"\n    \n    slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: int(x.InstanceNumber))\n    \n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed\ndef get_pixels_hu(scans):\n    \"\"\"\n    Converts raw images to Hounsfield Units (HU).\n    \n    Parameters: scans (Raw images)\n    \n    Returns: image (NumPy array)\n    \"\"\"\n    \n    image = np.stack([s.pixel_array for s in scans])\n    image = image.astype(np.int16)\n\n    # Since the scanning equipment is cylindrical in nature and image output is square,\n    # we set the out-of-scan pixels to 0\n    image[image == -2000] = 0\n    \n    \n    # HU = m*P + b\n    intercept = scans[0].RescaleIntercept\n    slope = scans[0].RescaleSlope\n    \n    if slope != 1:\n        image = slope * image.astype(np.float64)\n        image = image.astype(np.int16)\n        \n    image += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"INPUT_FOLDER = '/kaggle/input/osic-pulmonary-fibrosis-progression/train/'\n\npatients = os.listdir(INPUT_FOLDER)\npatients.sort()\ntest_patient_scans = load_scan(INPUT_FOLDER + patients[24])\ntest_patient_images = get_pixels_hu(test_patient_scans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_scans[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(test_patient_images[0], cmap='gray')\nplt.title(\"Original Slice\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed\ndef generate_markers(image):\n    \"\"\"\n    Generates markers for a given image.\n    \n    Parameters: image\n    \n    Returns: Internal Marker, External Marker, Watershed Marker\n    \"\"\"\n    \n    #Creation of the internal Marker\n    marker_internal = image < -400\n    marker_internal = segmentation.clear_border(marker_internal)\n    marker_internal_labels = measure.label(marker_internal)\n    \n    areas = [r.area for r in measure.regionprops(marker_internal_labels)]\n    areas.sort()\n    \n    if len(areas) > 2:\n        for region in measure.regionprops(marker_internal_labels):\n            if region.area < areas[-2]:\n                for coordinates in region.coords:                \n                       marker_internal_labels[coordinates[0], coordinates[1]] = 0\n    \n    marker_internal = marker_internal_labels > 0\n    \n    # Creation of the External Marker\n    external_a = ndimage.binary_dilation(marker_internal, iterations=10)\n    external_b = ndimage.binary_dilation(marker_internal, iterations=55)\n    marker_external = external_b ^ external_a\n    \n    # Creation of the Watershed Marker\n    marker_watershed = np.zeros((512, 512), dtype=np.int)\n    marker_watershed += marker_internal * 255\n    marker_watershed += marker_external * 128\n    \n    return marker_internal, marker_external, marker_watershed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed\ntest_patient_internal, test_patient_external, test_patient_watershed = generate_markers(test_patient_images[12])\n\nf, (ax1, ax2, ax3) = plt.subplots(1, 3, sharey=True, figsize=(15,15))\n\nax1.imshow(test_patient_internal, cmap='gray')\nax1.set_title(\"Internal Marker\")\nax1.axis('off')\n\nax2.imshow(test_patient_external, cmap='gray')\nax2.set_title(\"External Marker\")\nax2.axis('off')\n\nax3.imshow(test_patient_watershed, cmap='gray')\nax3.set_title(\"Watershed Marker\")\nax3.axis('off')\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# I like the internal marker. The area can be calculated by the number of \"white\" pixel multiple by the square of \"PixelSpacing\".","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_internal","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_internal.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_internal_array = np.ravel(test_patient_internal)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_internal_array.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"number_white_pixel = 0\nfor x in test_patient_internal_array:\n    if x == True:\n        number_white_pixel = number_white_pixel+1\nprint ('The number of white pixels is ', number_white_pixel)\nprint ('The percentage of white pixels in the dark image is {:0.2f}'.format( number_white_pixel/262144*100), '%')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# the pixel spacing of patient ID00062637202188654068490\nprint (Fore.BLUE + \"The pixel spacing of patient ID00062637202188654068490 is\", Style.RESET_ALL,test_patient_scans[0].PixelSpacing[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The area of chest in this slice\n\nprint (Fore.RED + \"The area of chest in the slice 12.dcm of patient ID00062637202188654068490 is\", Style.RESET_ALL,(test_patient_scans[0].PixelSpacing[0])*(test_patient_scans[0].PixelSpacing[0])*number_white_pixel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# only return the internal area\ndef only_internal(image):\n    \"\"\"\n    Generates markers for a given image.\n    \n    Parameters: image\n    \n    Returns: Internal Marker, External Marker, Watershed Marker\n    \"\"\"\n    \n    #Creation of the internal Marker\n    marker_internal = image < -400\n    marker_internal = segmentation.clear_border(marker_internal)\n    marker_internal_labels = measure.label(marker_internal)\n    \n    areas = [r.area for r in measure.regionprops(marker_internal_labels)]\n    areas.sort()\n    \n    if len(areas) > 2:\n        for region in measure.regionprops(marker_internal_labels):\n            if region.area < areas[-2]:\n                for coordinates in region.coords:                \n                       marker_internal_labels[coordinates[0], coordinates[1]] = 0\n    \n    marker_internal = marker_internal_labels > 0\n    \n    return marker_internal","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The volume of chest of the patient ID00062637202188654068490\nfor image in test_patient_images:\n    image_pixel = only_internal(image)\n    test_patient_internal_array = np.ravel(test_patient_internal)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}