{"cells":[{"metadata":{},"cell_type":"markdown","source":"# EDA (Exploritory Data Analysis)\n## OSIC Pulmonary Fibrosis Progression Competition","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install git+https://github.com/fastai/fastai2","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Packages","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom os import listdir\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport plotly.express as px\n\nfrom fastai2.basics           import *\nfrom fastai2.medical.imaging  import *\n\nimport pydicom\nimport matplotlib.pyplot as plot\n\nprint('Done!')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_dir = '../input/osic-pulmonary-fibrosis-progression'\nbase_path = Path('../input/osic-pulmonary-fibrosis-progression')\n\n#Can use DICOM metadata explored below (do for each patient) to look at average, max, and min deviation\n#of pixel lightness/darkness, find outliers that may be useful or not\ntrain_sample_path = base_path/'train'\n#.ls() is a PyPI method to replace default .dir() method\nfns_trn = base_path.ls()\ntrn_short = fns_trn[:5]\n\n#Work on printing the DICOM images using pyplot","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Are certain patients over or under-represented in the dataset? How many ","execution_count":null},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"df_trn = pd.DataFrame()\ndcm_per_patient = []\n\n#Gets all DICOM metadata\nfor folder in trn_short:\n    folder_contents = folder.ls()\n    num_of_dcms = len(folder_contents)\n    dcm_per_patient.append(num_of_dcms)\n    folder_df = pd.DataFrame.from_dicoms(folder_contents, px_summ=False)\n    df_trn = df_trn.append(folder_df, ignore_index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#total number of patients   \nprint('There are' + str(len(dcm_per_patient)) + ' total patients.')\n#mean\nprint(sum(dcm_per_patient)/len(dcm_per_patient))\n#median\nprint(dcm_per_patient[round(len(dcm_per_patient)/2)])\n\ndcm_plt = plt.figure(figsize = (16,8))\ndcm_subplt1 = plt.subplot(1, 2, 1, title='Distribution of DICOM files per patient', xlabel='Number of Files', ylabel='Number of Patients')\ndcm_subplt2 = plt.subplot(1, 2, 2, title='Patients with Less than 100 DICOM files', xlabel='Number of Files', ylabel='Number of Patients')\ndcm_subplt1.hist(dcm_per_patient)\ndcm_subplt2.hist(dcm_per_patient, bins=[0,10,20,30,40,50,60,70,80,90,100])\ndcm_subplt2.set_xlim([0,100])\n\nprint(\"There are x patients with less than 100 DICOM files.\")\nprint(\"There are x patients with greater than 500 DICOM files.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = Path('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntrain_csv = pd.read_csv(train_path)\ntrain_csv.pivot_table(index='Patient')\n\npatient_ids = train_csv['Patient'].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"categorical_df = pd.DataFrame()\nfor id_number in patient_ids:\n    patient_info = train_csv.loc[train_csv['Patient'] == id_number]\n    patient_sex = patient_info.iloc[0]\n    categorical_df = categorical_df.append(patient_sex)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.sunburst(\n    categorical_df,\n    path=['Sex','SmokingStatus'],\n    color_discrete_sequence=[\"#247BA0\", \"#70C1B3\"],\n    title='Percentage of Patients by Sex and Smoking Status')\nfig.update_traces(textinfo='label+percent parent')\nfig.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_scans(dcm_path):\n    slices = [pydicom.dcmread(dcm_path + \"/\" + file) for file in listdir(dcm_path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2]))\n    return slices\n\nsample = base_dir + '/train/' + patient_ids[0]\ndcms = load_scans(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1,4,figsize=(20,3))\nax[0].set_title(\"Original CT-scan\")\nax[0].imshow(dcms[0].pixel_array, cmap=\"bone\")\nax[1].set_title(\"Pixelarray distribution\")\nax[1].hist(dcms[0].pixel_array.flatten())","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}