{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nfrom matplotlib import pyplot as plt\nimport pydicom\nimport os\nimport random\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Number of unique patients"},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train['Patient'].unique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of patients visit"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(train.groupby(['Patient']).count()['Weeks'].to_list())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Age Distribution in train dataset[](http://)"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n#fig = plt.Figure(figsize=(28,42),constrained_layout=True)\nfig, axs = plt.subplots(2, 2,figsize=(15,10))\naxs[0, 0].hist(train['Age'])\naxs[0, 0].set_title('Age Distribution')\naxs[0, 1].hist(train['SmokingStatus'])\naxs[0, 1].set_title('Smokers Distribution')\naxs[1, 0].hist(train['Sex'])\naxs[1, 0].set_title('Gender Distribution')\naxs[1, 1].hist(train['Weeks'],bins=50)\naxs[1, 1].set_title('Weeks Distribution')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Percent distribution across different age groups"},{"metadata":{"trusted":true},"cell_type":"code","source":"width = (max(train['Weeks'])-min(train['Weeks']))/4\nrow = 2\ncol = 2\n_min = min(train['Weeks'])\ndata = dict()\nfor i in range(0,row):\n    for j in range(0,col):\n        c_graph = i*col+j\n        llimit = _min+c_graph*width\n        ulimit = llimit+width\n        key = 'Weeks '+str(llimit)+' - '+str(ulimit)\n        data[key] = train[(train['Weeks']<=ulimit) & (train['Weeks']>llimit)]['Percent']\nfig, ax = plt.subplots(figsize=(15,10))\nax.boxplot(data.values())\nax.set_title('Distibution of percent with Week #')\nax.set_xticklabels(data.keys())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## FVC Distribution across different age group"},{"metadata":{"trusted":true},"cell_type":"code","source":"width = (max(train['Age'])-min(train['Age']))/4\nrow = 2\ncol = 2\n_min = min(train['Age'])\ndata = dict()\nfor i in range(0,row):\n    for j in range(0,col):\n        c_graph = i*col+j\n        llimit = _min+c_graph*width\n        ulimit = llimit+width\n        key = 'Age '+str(llimit)+' - '+str(ulimit)\n        data[key] = train[(train['Age']<=ulimit) & (train['Age']>llimit)]['FVC']\nfig, ax = plt.subplots(figsize=(15,10))\nax.boxplot(data.values())\nax.set_title('Distibution of FVC with Patient\\'s age')\nax.set_xticklabels(data.keys())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"width = (max(train['Age'])-min(train['Age']))/4\nrow = 2\ncol = 2\n_min = min(train['Age'])\ndata = dict()\nfor i in range(0,row):\n    for j in range(0,col):\n        c_graph = i*col+j\n        llimit = _min+c_graph*width\n        ulimit = llimit+width\n        key = 'Age '+str(llimit)+' - '+str(ulimit)\n        data[key] = train[(train['Age']<=ulimit) & (train['Age']>llimit)]['Percent']\nfig, ax = plt.subplots(figsize=(15,10))\nax.boxplot(data.values())\nax.set_title('Distibution of percent with Patient\\'s Age')\nax.set_xticklabels(data.keys())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['SmokingStatus'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = dict()\nsmokingstatus = train['SmokingStatus'].unique()\nfor i in range(len(smokingstatus)):\n    key = smokingstatus[i]\n    data[key] = train[train['SmokingStatus']==key]['Percent']\nfig, ax = plt.subplots(figsize=(15,10))\nax.boxplot(data.values())\nax.set_title('Distibution of percent with Patient\\'s Smoking Status')\nax.set_xticklabels(data.keys())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = dict()\ngenders = train['Sex'].unique()\nfor i in range(len(genders)):\n    key = genders[i]\n    data[key] = train[train['Sex']==key]['Percent']\nfig, ax = plt.subplots(figsize=(15,10))\nax.boxplot(data.values())\nax.set_title('Distibution of percent with Patient\\'s Gender')\nax.set_xticklabels(data.keys())\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keys = ['BitsAllocated','BitsStored',\n 'BodyPartExamined',\n 'Columns',\n 'ConvolutionKernel',\n 'DeidentificationMethod',\n 'DistanceSourceToDetector',\n 'DistanceSourceToPatient',\n 'FocalSpots',\n 'FrameOfReferenceUID',\n 'GantryDetectorTilt',\n 'GeneratorPower',\n 'HighBit',\n 'ImageOrientationPatient',\n 'ImagePositionPatient',\n 'ImageType',\n 'InstanceNumber',\n 'KVP',\n 'Manufacturer',\n 'ManufacturerModelName',\n 'Modality',\n 'PatientID',\n 'PatientName',\n 'PatientPosition',\n 'PatientSex',\n 'PhotometricInterpretation',\n 'PixelData',\n 'PixelRepresentation',\n 'PixelSpacing',\n 'PositionReferenceIndicator',\n 'RescaleIntercept',\n 'RescaleSlope',\n 'RotationDirection',\n 'Rows',\n 'SOPInstanceUID',\n 'SamplesPerPixel',\n 'SeriesInstanceUID',\n 'SliceLocation',\n 'SliceThickness',\n 'StudyID',\n 'StudyInstanceUID',\n 'TableHeight',\n 'WindowCenter',\n 'WindowCenterWidthExplanation',\n 'WindowWidth',\n 'XRayTubeCurrent']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_image_meta = dict()\nfrom datetime import datetime as dt\nfrom tqdm import tqdm\nfor key in keys:\n    all_image_meta[key] = list()\nBreak = False\nfor _file in tqdm(os.listdir('/kaggle/input/osic-pulmonary-fibrosis-progression/train/')):\n    for _img in os.listdir('/kaggle/input/osic-pulmonary-fibrosis-progression/train/'+_file):\n        dataset = pydicom.dcmread('/kaggle/input/osic-pulmonary-fibrosis-progression/train/'+_file+'/'+_img)\n        all_image_meta['file_path'] = '/kaggle/input/osic-pulmonary-fibrosis-progression/train/'+ _file + '/' + _img\n        for key in keys:\n            try:\n                all_image_meta[key].append(str(dataset[key]).split(' ')[-1])\n            except KeyError as k:\n                all_image_meta[key].append(None)\n            except Exception as e:\n                Break = True\n                break\n        if Break:\n            break\n    if Break:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_image_meta = pd.DataFrame(all_image_meta)\nall_image_meta.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Look at a sample for each column type and number of uniques values in column"},{"metadata":{"trusted":true},"cell_type":"code","source":"for key in keys:\n    print(key,all_image_meta[key].nunique(),all_image_meta[key][0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_patient_id(_str):\n    return _str.replace('\\'','')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_image_meta['PatientID'] = all_image_meta['PatientID'].apply(clean_patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Check the CT Scan Progression of a sample patient from each category"},{"metadata":{},"cell_type":"markdown","source":"### Average number of file per patient"},{"metadata":{"trusted":true},"cell_type":"code","source":"all_image_meta.groupby('PatientID').count()['Columns'].mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def view_patient(patient_id):\n    # Draw random 15 images for the patient\n    files = os.listdir('/kaggle/input/osic-pulmonary-fibrosis-progression/train/'+patient_id)\n    files = random.sample(files, 15)\n    \n    # Draw the images\n    row = 3\n    col = 5\n    fig, axs = plt.subplots(row, col,figsize=(15,10))\n    for i in range(0,row):\n        for j in range(0,col):\n            c_image = files[i*col+j]\n            data = pydicom.dcmread('/kaggle/input/osic-pulmonary-fibrosis-progression/train/'+patient_id+'/'+c_image)\n            axs[i, j].imshow(data.pixel_array, cmap=plt.cm.bone)\n            axs[i,j].set_title(c_image)\n    plt.show()\n    \n    # plot the FVC and percentage progression for the patient\n    data = train[train['Patient'] == patient_id]\n    plt.plot(data['Weeks'], data['FVC'],color = 'red')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualize a patient with high FVC and age 49-55"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['Percent']>80) & (train['Age'] < 55)]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualize a patient with high FVC and age 55-65"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['Percent']>80) & (train['Age'] > 55)]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Visualize a patient with low percent and age 45 - 55"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['Percent']<60) & (train['Age'] < 55)]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Visualize a patient with low percent and age >55"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['Percent']<60) & (train['Age'] > 55)]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Graph for a smoker"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['SmokingStatus'] == 'Currently smokes')]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Graph for a non-smoker"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['SmokingStatus'] == 'Never smoked')]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### progression graph for ex smokers"},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_id = np.random.choice(train[(train['SmokingStatus'] == 'Ex-smoker')]['Patient'].to_list())\nview_patient(patient_id)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}