{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        pass\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\npd.plotting.register_matplotlib_converters()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"file_path_train='../input/osic-pulmonary-fibrosis-progression/train.csv'\nraw_data =pd.read_csv(file_path_train)\nraw_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_data.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Smoking Status","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# reading the data from the first week of evry patient\ndf=raw_data.groupby(['Patient']).first()\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('The Totl number of patients visited :',len(raw_data.Patient.unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Smoke=df.groupby(['SmokingStatus']).count()['Sex'].to_frame()\nSmoke","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.barplot(x=Smoke.Sex.keys(),y=Smoke.Sex.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.groupby(['Sex']).count()['SmokingStatus'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nsns.countplot(data=df, x='SmokingStatus', hue='Sex')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## In-sights\n`1.There are more number of patients belongs to ex-smokers and currently smoking patients are very less.\n2.From gender prespective many patients are men` ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Age distirbution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Age dostribution plot \nmu=df.Age.std()\nmean=df.Age.mean()\nplt.figure(figsize=(10,6))\nplt.title('Age distirbution [mu {:.2f} and mean {:.2f}]'.format(mu,mean),fontsize=15,color='black')\nsns.distplot(df['Age'],kde=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# smoking staus versus Age distribution\nsmoker_dist=df.loc[df.SmokingStatus=='Currently smokes']['Age']\nexsmoker_dist=df.loc[df.SmokingStatus=='Ex-smoker']['Age']\nnonsmoker_dist=df.loc[df.SmokingStatus=='Never smoked']['Age']\n\nplt.figure(figsize=(10,6))\nsns.kdeplot(smoker_dist,shade=True,label='currenty smokes')\nsns.kdeplot(exsmoker_dist,shade=True,label='Ex-smoker')\nsns.kdeplot(nonsmoker_dist,shade=True,label='Never smoked')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Gender and Age distribution\nMale_dist=df.loc[df.Sex=='Male']['Age']\nFemale_dist=df.loc[df.Sex=='Female']['Age']\n\nplt.figure(figsize=(10,6))\nsns.kdeplot(Male_dist,shade=True,label='Male')\nsns.kdeplot(Female_dist,shade=True,label='Female')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.swarmplot(x=df[\"Sex\"],y=df['Age'],hue=df['SmokingStatus'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# FVC and Percentage","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_ids=raw_data.Patient.unique()\n\npatient_week=[]\npatient_fvc=[]\npatient_percentage=[]\nfor ids in patient_ids:\n    week=raw_data.loc[raw_data['Patient']==ids]['Weeks'].values\n    fvc=raw_data.loc[raw_data['Patient']==ids]['FVC'].values\n    percent=raw_data.loc[raw_data['Patient']==ids]['Percent'].values\n    patient_week.append(week)\n    patient_fvc.append(fvc)\n    patient_percentage.append(percent)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.title(\"Each patient's FVC decay over the weeks\")\nplt.xlabel('Weeks')\nplt.ylabel('FVC deacy ')\nfor i in range(len(patient_ids)):\n    sns.lineplot(x=patient_week[i],y=patient_fvc[i],label ='P'+str(i+1),lw=1,legend='full')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.title(\"Each patient's Percentage over the weeks\")\nplt.xlabel('Weeks')\nplt.ylabel('Percentage')\nfor i in range(len(patient_ids)):\n    sns.lineplot(x=patient_week[i],y=patient_percentage[i],label ='P'+str(i+1),lw=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.title(\"Each patient's Percentage Vs FVC\")\nplt.xlabel('FVC')\nplt.ylabel('Percentage')\nfor i in range(len(patient_ids)):\n    sns.lineplot(x=patient_fvc[i],y=patient_percentage[i],label ='P'+str(i+1),lw=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}