{"cells":[{"metadata":{},"cell_type":"markdown","source":"<img src='https://www.osicild.org/uploads/1/2/2/7/122798879/editor/kaggle-v01-clipped.png?1569346633'>\n<h1><center>OSIC Pulmonary Fibrosis Progression - EDA</center><h1>\n    \n# 1. Introduction ▶\n    \n###  1.1 What is Pulmonary fibrosis?\n* [Pulmonary fibrosis is a lung disease that occurs when lung tissue becomes damaged and scarred.](https://www.mayoclinic.org/diseases-conditions/pulmonary-fibrosis/symptoms-causes/syc-20353690)  This thickened, stiff tissue makes it more difficult for your lungs to work properly.\n    \n###  1.2 What we need to do? Obervation\nWe will predict a patient’s severity of decline in lung function based on a CT scan of their lungs.\n    \n- This leaderboard is calculated with approximately 1% of the test data. The final results will be based on the other 99%, so the final standings may be different.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# 1. Importing the necessary libraries ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nfrom os import listdir\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n!pip install dicom\nimport dicom\nimport os\nimport numpy\nfrom matplotlib import pyplot, cm\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\nimport seaborn as sns\nsns.set(style=\"whitegrid\")\n\n\n#pydicom\nimport pydicom\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')\n\n\n# Settings for pretty nice plots\nplt.style.use('fivethirtyeight')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PathDicom = \"../input/osic-pulmonary-fibrosis-progression/train/ID00422637202311677017371/\"\nlstFilesDCM = []  # create an empty list\nfor dirName, subdirList, fileList in os.walk(PathDicom):\n    for filename in fileList:\n        if \".dcm\" in filename.lower():  # check whether the file's DICOM\n            lstFilesDCM.append(os.path.join(dirName,filename))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install natsort\nimport natsort\n# print(natsort.natsorted(lstFilesDCM,reverse=False))\nlstFilesDCM = natsort.natsorted(lstFilesDCM,reverse=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom as dicom\nimport PIL # optional\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# specify your image path\n#image_path = 'xray.dcm'\nPathDicom = \"../input/osic-pulmonary-fibrosis-progression/train/\"\nlist_patients = [x[0] for x in os.walk(PathDicom)]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for patient in list_patients:\n    lstFilesDCM = []  # create an empty list\n    for dirName, subdirList, fileList in os.walk(patient):\n        for filename in fileList:\n            if \".dcm\" in filename.lower():  # check whether the file's DICOM\n                lstFilesDCM.append(os.path.join(dirName,filename))\n\n    lstFilesDCM = natsort.natsorted(lstFilesDCM,reverse=False)\n\n    for i in range(len(lstFilesDCM)):\n        ds = dicom.dcmread(lstFilesDCM[i])\n        print(ds)\n        plt.imshow(ds.pixel_array)\n        plt.show()\n        break\n    break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. Reading the train.csv","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# List files available\nlist(os.listdir(\"../input/osic-pulmonary-fibrosis-progression\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Defining data path\nIMAGE_PATH = \"../input/osic-pulmonary-fibrosis-progressiont/\"\n\ntrain_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\n\n\n#Training data\nprint('Training data shape: ', train_df.shape)\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df[\"typical_FVC\"] = (train_df[\"FVC\"]*100)/train_df[\"Percent\"]\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure()\nplt.plot(train_df[\"Weeks\"], train_df[\"FVC\"], \"o\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure()\nplt.plot(train_df[\"Weeks\"], train_df[\"Percent\"], \"o\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn import linear_model\nimport statsmodels.api as sm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0\nfor pid,tdf in train_df.groupby(\"Patient\"):\n    if i % 10 == 0:\n        sns.lmplot(x='Weeks',y='Percent',data=tdf,fit_reg=True)\n        regr = linear_model.LinearRegression()\n        X = tdf.Weeks.values.reshape(-1,1)\n        y = tdf.Percent.values.reshape(-1,1)\n        regr.fit(X, y)\n        print(regr.coef_[0], regr.intercept_)\n    i += 1\nprint(i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0\nfor pid,tdf in train_df.groupby(\"Patient\"):\n    if i % 10 == 0:\n        sns.lmplot(x='Weeks',y='Percent',data=tdf,fit_reg=True)\n        X = tdf.Weeks.values.reshape(-1,1)\n        X = sm.add_constant(X)\n        y = tdf.Percent.values.reshape(-1,1)\n        model = sm.OLS(y,X)\n        results = model.fit()\n        print(results.params)\n        print(results.bse)\n    i += 1\nprint(i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i = 0\ndf = {}\ndf[\"Patient\"] = []\ndf[\"slope\"] = []\ndf[\"bse\"] = []\nfor pid,tdf in train_df.groupby(\"Patient\"):\n    X = tdf.Weeks.values.reshape(-1,1)\n    X = sm.add_constant(X)\n    y = tdf.Percent.values.reshape(-1,1)\n    model = sm.OLS(y,X)\n    results = model.fit()\n    df[\"Patient\"].append(pid)\n    df[\"slope\"].append(results.params[1])\n    df[\"bse\"].append(results.bse[1])\n    i += 1\nprint(i)\ndf = pd.DataFrame(df)\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"typ_fvc_df = train_df.groupby(['Age', 'Sex', 'SmokingStatus']).mean()['typical_FVC'].to_frame().reset_index()\ntyp_fvc_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"typ_fvc_df.groupby(['Sex', 'SmokingStatus']).mean()['typical_FVC'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"conditions = [\n    (typ_fvc_df['Age'] <= 50),\n    (typ_fvc_df['Age'] > 50) & (typ_fvc_df['Age'] <= 60),\n    (typ_fvc_df['Age'] > 60) & (typ_fvc_df['Age'] <= 70),\n    (typ_fvc_df['Age'] > 70) & (typ_fvc_df['Age'] <= 80)]\nchoices = [0,1,2,3]\ntyp_fvc_df['age_group'] = np.select(conditions, choices, default=4)\ntyp_fvc_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"typ_fvc_df.groupby(['Sex', 'SmokingStatus', 'age_group']).mean()['typical_FVC'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df.groupby(['SmokingStatus']).count()['Sex'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3. Data Exploration","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Null values and Data types\nprint('Train Set !!')\nprint(train_df.info())\nprint('-------------')\nprint('Test Set !!')\nprint(test_df.info())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Total number of Patient in the dataset(train+test)\nprint(\"Total Patient in Train set: \",train_df['Patient'].count())\nprint(\"Total Patient in Test set: \",test_df['Patient'].count())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Unique Patients(Ids)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"The total patient ids are {train_df['Patient'].count()}, from those the unique ids are {train_df['Patient'].value_counts().shape[0]} \")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns = train_df.keys()\ncolumns = list(columns)\nprint(columns)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Exploring the 'SmokingStatus' column","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['SmokingStatus'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['SmokingStatus'].value_counts(normalize=True).iplot(kind='bar',\n                                                      yTitle='Percentage', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='red',\n                                                      theme='pearl',\n                                                      bargap=0.8,\n                                                      gridcolor='white',\n                                                     \n                                                      title='Distribution of the SmokingStatus column in the training set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Weeks distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Weeks'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Weeks'].value_counts().sort_values().iplot(kind='barh',\n                                                      xTitle='Counts(Weeks)', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='#FB8072',\n                                                      theme='pearl',\n                                                      bargap=0.2,\n                                                      gridcolor='white',\n                                                      title='Distribution of the Weeks in the training set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Weeks vs SmokingStatus","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"z=train_df.groupby(['SmokingStatus','Weeks'])['FVC'].count().to_frame().reset_index()\nz.style.background_gradient(cmap='Reds') ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## FVC","execution_count":null},{"metadata":{},"cell_type":"markdown","source":" The forced vital capacity (FVC), i.e. the volume of air exhaled\n - the recorded lung capacity in ml","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['FVC'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['FVC'].value_counts().iplot(kind='barh',\n                                      xTitle='Lung Capacity(ml)', \n                                      linecolor='black', \n                                      opacity=0.7,\n                                      color='#FB8072',\n                                      #|theme='pearl',\n                                      bargap=0.5,\n                                      gridcolor='white',\n                                      title='Distribution of the FVC in the training set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Percent","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"A computed field which approximates the patient's FVC as a percent of the typical FVC for a person of similar characteristics","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Percent'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Percent'].iplot(kind='hist',bins=30,color='blue',xTitle='Percent distribution',yTitle='Count')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Age Distribution of patients","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Age'].iplot(kind='hist',bins=30,color='red',xTitle='Age distribution',yTitle='Count')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Distribution of Age vs SmokingStatus","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['SmokingStatus'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.kdeplot(train_df.loc[train_df['SmokingStatus'] == 'Ex-smoker', 'Age'], label = 'Ex-smoker',shade=True)\n\nsns.kdeplot(train_df.loc[train_df['SmokingStatus'] == 'Never smoked', 'Age'], label = 'Never smoked',shade=True)\n\nsns.kdeplot(train_df.loc[train_df['SmokingStatus'] == 'Currently smokes', 'Age'], label = 'Currently smokes',shade=True)\n\n# Labeling of plot\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Distribution of Age vs gender","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.kdeplot(train_df.loc[train_df['Sex'] == 'Male', 'Age'], label = 'Male',shade=True)\n\nsns.kdeplot(train_df.loc[train_df['Sex'] == 'Female', 'Age'], label = 'Female',shade=True)\n\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Gender distribution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Sex'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['Sex'].value_counts().iplot(kind='bar',\n                                          yTitle='Percentage', \n                                          linecolor='black', \n                                          opacity=0.7,\n                                          color='blue',\n                                          theme='pearl',\n                                          bargap=0.8,\n                                          gridcolor='white',\n\n                                          title='Distribution of the Sex column in the training set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Gender vs SmokingStatus","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\na = sns.countplot(data=train_df, x='SmokingStatus', hue='Sex')\n\nfor p in a.patches:\n    a.annotate(format(p.get_height(), ','), \n           (p.get_x() + p.get_width() / 2., \n            p.get_height()), ha = 'center', va = 'center', \n           xytext = (0, 4), textcoords = 'offset points')\n\nplt.title('Gender split by SmokingStatus', fontsize=16)\nsns.despine(left=True, bottom=True);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from IPython.display import Image\nImage(filename='../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/1.dcm')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}