{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\npd.plotting.register_matplotlib_converters()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_path='../input/osic-pulmonary-fibrosis-progression'\nraw_data =pd.read_csv(file_path+'/train.csv')\ntest_data =pd.read_csv(file_path+'/test.csv')\nsub =pd.read_csv(file_path+'/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('The shape of the training dataset is ', raw_data.shape)\nprint('The shape of the training dataset is ', test_data.shape)\nprint('The Totl number of patients visited :',len(raw_data.Patient.unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_data.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Smoking Status","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# reading the data from the first week of evry patient\ndf=raw_data.groupby(['Patient']).first()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Smoke=df.groupby(['SmokingStatus']).count()['Sex'].to_frame()\nSmoke","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.barplot(x=Smoke.Sex.keys(),y=Smoke.Sex.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.groupby(['Sex']).count()['SmokingStatus'].to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nsns.countplot(data=df, x='SmokingStatus', hue='Sex')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## In-sights\n`1.There are more number of patients belongs to ex-smokers and currently smoking patients are very less.\n2.From gender prespective many patients are men` ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Age distirbution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Age dostribution plot \nmu=df.Age.std()\nmean=df.Age.mean()\nplt.figure(figsize=(10,6))\nplt.title('Age distirbution [mu {:.2f} and mean {:.2f}]'.format(mu,mean),fontsize=15,color='black')\nsns.distplot(df['Age'],kde=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# smoking staus versus Age distribution\nsmoker_dist=df.loc[df.SmokingStatus=='Currently smokes']['Age']\nexsmoker_dist=df.loc[df.SmokingStatus=='Ex-smoker']['Age']\nnonsmoker_dist=df.loc[df.SmokingStatus=='Never smoked']['Age']\n\nplt.figure(figsize=(10,6))\nsns.kdeplot(smoker_dist,shade=True,label='currenty smokes')\nsns.kdeplot(exsmoker_dist,shade=True,label='Ex-smoker')\nsns.kdeplot(nonsmoker_dist,shade=True,label='Never smoked')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Gender and Age distribution\nMale_dist=df.loc[df.Sex=='Male']['Age']\nFemale_dist=df.loc[df.Sex=='Female']['Age']\n\nplt.figure(figsize=(10,6))\nsns.kdeplot(Male_dist,shade=True,label='Male')\nsns.kdeplot(Female_dist,shade=True,label='Female')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.swarmplot(x=df[\"Sex\"],y=df['Age'],hue=df['SmokingStatus'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# FVC and Percentage","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_ids=raw_data.Patient.unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_week=[]\npatient_fvc=[]\npatient_percentage=[]\nfor ids in patient_ids:\n    week=raw_data.loc[raw_data['Patient']==ids]['Weeks'].values\n    fvc=raw_data.loc[raw_data['Patient']==ids]['FVC'].values\n    percent=raw_data.loc[raw_data['Patient']==ids]['Percent'].values\n    patient_week.append(week)\n    patient_fvc.append(fvc)\n    patient_percentage.append(percent)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.title(\"Each patient's FVC decay over the weeks\")\nplt.xlabel('Weeks')\nplt.ylabel('FVC deacy ')\nfor i in range(len(patient_ids)):\n    sns.lineplot(x=patient_week[i],y=patient_fvc[i],label ='P'+str(i+1),lw=1,legend=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.title(\"Each patient's Percentage over the weeks\")\nplt.xlabel('Weeks')\nplt.ylabel('Percentage')\nfor i in range(len(patient_ids)):\n    sns.lineplot(x=patient_week[i],y=patient_percentage[i],label ='P'+str(i+1),lw=1,legend=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.title(\"Each patient's Percentage Vs FVC\")\nplt.xlabel('FVC')\nplt.ylabel('Percentage')\nfor i in range(len(patient_ids)):\n    sns.lineplot(x=patient_fvc[i],y=patient_percentage[i],label ='P'+str(i+1),lw=1,legend=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training Set up","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_base = raw_data.drop_duplicates(subset='Patient', keep='first')\ndf_base = df_base[['Patient', 'Weeks', 'FVC', \n                   'Percent', 'Age']].rename(columns={'Weeks': 'base_week',\n                                                      'Percent': 'base_percent',\n                                                      'Age': 'base_age',\n                                                      'FVC': 'base_FVC'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_train =raw_data.merge(df_base,how='left',on=['Patient'])\ndata_train =data_train.loc[data_train.Weeks!=data_train.base_week]# removing the first week from the weeks\ndata_train['week_count']=data_train.Weeks-data_train.base_week # to check the weeks count from base week\n\ndata_train= pd.get_dummies(data_train,columns=['Sex','SmokingStatus']) # to get the dummy columns for Sex and smokingststaus","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_train_inp_file =data_train.drop(columns=['Patient','FVC','Percent','Weeks','Age'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target =data_train['FVC']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Metrics","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def log_likely_hood(y_true,y_pred,y_pred_std):\n    \n    sigma_clipped = np.maximum(y_pred_std,70)\n    \n    delta = np.minimum(abs(y_true-y_pred),1000)\n    \n    metric = -(np.sqrt(2*delta)/sigma_clipped)-np.log(np.sqrt(2*sigma_clipped))\n    \n    return np.mean(metric)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.linear_model import ARDRegression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(data_train_inp_file,target,test_size=0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train.shape,y_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscalar= StandardScaler()\n\nX_train_scaled = scalar.fit_transform(X_train)\n\nX_test_scaled = scalar.fit_transform(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ard= ARDRegression()\nard.fit(X_train_scaled,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred,y_pred_std=ard.predict(X_test_scaled,return_std=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"log_likely_hood(y_test,y_pred,y_pred_std)# prediction on training set","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# preparing test data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sub['Patient'] = sub['Patient_Week'].apply(lambda x: x.split('_')[0])\nsub['Weeks'] = sub['Patient_Week'].apply(lambda x: x.split('_')[1]).astype(int)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_mod = sub.drop(columns=['FVC','Confidence'],axis=1)\nsub_mod.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test=test_data.rename(columns={'Weeks': 'base_week',\n                                'Percent': 'base_percent',\n                                'Age': 'base_age',\n                                'FVC': 'base_FVC'})\n\ndf_test=pd.get_dummies(df_test,columns=['Sex','SmokingStatus'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test['Sex_Female']=0\ndf_test['SmokingStatus_Currently smokes']=0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test_mod2=sub_mod.merge(df_test,how='left',on=['Patient'])\nsub2= df_test_mod2.copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub2.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test_inp_file = sub2.copy()\ndata_test_inp_file['week_count']= data_test_inp_file.Weeks-data_test_inp_file.base_week\ndata_test_inp_file.drop(['Patient','Weeks'],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features=data_train_inp_file.columns\ndata_test_inp_file_id=data_test_inp_file['Patient_Week']\ndata_test_file =data_test_inp_file.drop(['Patient_Week'],axis=1)\ndata_test_file = data_test_file[features]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_test_file.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred_test,y_pred_test_std=ard.predict(data_test_file.values,return_std=True)\nsubmission=pd.DataFrame({'Patient_Week':data_test_inp_file_id,'FVC':y_pred_test,'Confidence':y_pred_test_std})\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}