{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport pymc3 as pm\nfrom sklearn import preprocessing\nfrom sklearn import datasets, linear_model\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\n\npd.set_option('display.max_rows', 176)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"scrolled":false},"cell_type":"code","source":"train = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/train.csv\")\n\nnum_columns = [\"Weeks\", \"FVC\", \"Percent\", \"Age\"]\ncat_columns = [\"Sex\", \"SmokingStatus\"]\n\nprint(train.columns)\nprint()\nprint(train[num_columns].describe())\nprint()\nprint(train[cat_columns].describe())\nprint()\nprint(train[\"Sex\"].unique())\nprint()\nprint(train[\"SmokingStatus\"].unique())\nprint()\ntrain.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"train_patient_summary = (train.groupby(\"Patient\")['Weeks', 'FVC', 'Percent'].agg([len, max, min]))\n\nprint(train_patient_summary.describe())\n\nprint(train_patient_summary.columns)\n\ntrain_patient_summary.head(176)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train_check_age = train.groupby(\"Patient\")[\"Age\"].nunique()\n#train_check_age.where(train_check_age > 1)\ntrain_check_age[train_check_age != 1].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train_check_sex = train.groupby(\"Patient\")[\"Sex\"].nunique()\n#train_check_age.where(train_check_age > 1)\ntrain_check_sex[train_check_sex != 1].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train_check_smoke = train.groupby(\"Patient\")[\"SmokingStatus\"].nunique()\n#train_check_age.where(train_check_age > 1)\ntrain_check_smoke[train_check_smoke != 1].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train[train[\"Patient\"] == 'ID00007637202177411956430'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"#patient_ids = [\"ID00007637202177411956430\", \"ID00009637202177434476278\", \"ID00007637202177411956430\", \"ID00009637202177434476278\"]\n\npatient_ids = train[\"Patient\"].unique()\n\ni = 0\nfig = plt.figure(figsize=(30,900))\nfig.subplots_adjust(hspace=0.5, wspace=0.5)\nsns.set_context(\"talk\")\n\nfor patient_id in patient_ids:\n    i += 1\n    train_patient = train.loc[train.Patient == patient_id]\n    fig.add_subplot(90,2,i)\n    sns.regplot(x=train_patient[\"Weeks\"], y=train_patient[\"Percent\"], ci=None).set_title(patient_id, fontsize=50)\n    \n\n#train[\"Patient\"].unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"#train_patient_summary[('Weeks', 'max')] \n#train_patient_summary[('Weeks', 'min')]\n\ntrain_patient_summary[('Weeks', 'max')] - train_patient_summary[('Weeks', 'min')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"train_patient_summary[('FVC', 'max')] - train_patient_summary[('FVC', 'min')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"train_patient_summary[('Percent', 'max')] - train_patient_summary[('Percent', 'min')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train_simple = train.groupby('Patient')['Age', 'Sex', 'SmokingStatus'].agg(['max'])\ntrain_simple.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train_compact = train_simple.droplevel(1,axis=1)\ntrain_compact.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"# Diffs across weeks\ntrain_compact = train_compact.assign(time_period=train_patient_summary[('Weeks', 'max')] - train_patient_summary[('Weeks', 'min')])\ntrain_compact = train_compact.assign(FVC_diff=train_patient_summary[('FVC', 'max')] - train_patient_summary[('FVC', 'min')])\ntrain_compact = train_compact.assign(Percent_diff=train_patient_summary[('Percent', 'max')] - train_patient_summary[('Percent', 'min')])\n\n# Average rates across weeks\ntrain_compact = train_compact.assign(FVC_rate=train_compact.FVC_diff / train_compact.time_period)\ntrain_compact = train_compact.assign(Percent_rate=train_compact.Percent_diff / train_compact.time_period)\n\n# Values at first week\ntrain_compact = train_compact.assign(First_week=train.sort_values(by='Weeks').groupby('Patient').first().Weeks)\ntrain_compact = train_compact.assign(First_FVC=train.sort_values(by='Weeks').groupby('Patient').first().FVC)\ntrain_compact = train_compact.assign(First_Percent=train.sort_values(by='Weeks').groupby('Patient').first().Percent)\n\n\n#train_compact\n#sns.kdeplot(train_compact['time_period'])\ntrain_compact.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Feeding numerical, categorical, image data into 1 neural network:  \nhttps://www.pyimagesearch.com/2019/01/21/regression-with-keras/  \nhttps://www.pyimagesearch.com/2019/01/28/keras-regression-and-cnns/  \nhttps://www.pyimagesearch.com/2019/02/04/keras-multiple-inputs-and-mixed-data/"},{"metadata":{"trusted":true,"scrolled":false},"cell_type":"code","source":"train = pd.read_csv(\"../input/osic-pulmonary-fibrosis-progression/train.csv\")\n# Very simple pre-processing: adding patient class\ndef patient_class(row):\n    if row['Sex'] == 'Male':\n        if row['SmokingStatus'] == 'Currently smokes':\n            return 0\n        elif row['SmokingStatus'] == 'Ex-smoker':\n            return 1\n        elif row['SmokingStatus'] == 'Never smoked':\n            return 2\n    else:\n        if row['SmokingStatus'] == 'Currently smokes':\n            return 3\n        elif row['SmokingStatus'] == 'Ex-smoker':\n            return 4\n        elif row['SmokingStatus'] == 'Never smoked':\n            return 5\n\ntrain['Class'] = train.apply(patient_class, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# Very simple pre-processing: adding FVC and week baselines\naux = train[['Patient', 'Weeks']].groupby('Patient')\\\n    .min().reset_index()\naux = pd.merge(aux, train[['Patient', 'Weeks', 'FVC', 'Percent']], how='left', \n               on=['Patient', 'Weeks'])\naux = aux.groupby('Patient').mean().reset_index()\naux['Weeks'] = aux['Weeks'].astype(int)\naux['FVC'] = aux['FVC'].astype(int)\naux['Percent'] = aux['Percent'].astype(float)\ntrain = pd.merge(train, aux, how='left', on='Patient', suffixes=('', '_base'))\ntrain['FVC_nom'] = train['FVC']*(100./train['Percent'])\ntrain.head(30)\n#train_fvc_nom = train.groupby(\"Patient\")[\"FVC_nom\"].max() - train.groupby(\"Patient\")[\"FVC_nom\"].min()\n#train_fvc_nom.max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patients = pd.DataFrame()\n# Very simple pre-processing: creating patient indexes\nle = preprocessing.LabelEncoder()\ntrain['PatientID'] = le.fit_transform(train['Patient'])\n\npatients = train[['Patient', 'PatientID', 'Age', 'Class', 'Weeks_base', 'FVC_base', 'Percent_base']].drop_duplicates()\nfvc_data = train[['Patient', 'PatientID', 'Weeks', 'FVC']]\npatients['FVC_nom_base'] = patients['FVC_base']*(100./patients['Percent_base'])\npatients.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\npatient_id = 'ID00007637202177411956430'\ndf_patient = pd.DataFrame(train.loc[train.Patient == patient_id])\nx = np.array(train.loc[train.Patient == patient_id, 'Weeks'].values).reshape(-1,1)\nx\ny = train.loc[train.Patient == patient_id, 'FVC'].values\n#y.reshape(-1,1)\nregr = linear_model.LinearRegression()\nregr.fit(x,y)\n\nprint('Gradient: \\n', regr.coef_[0])\nprint('Intercept: \\n', regr.intercept_)\n\ny_pred = regr.predict(x)\nprint('Mean absolute error: %.2f'% mean_absolute_error(y, y_pred))\nprint('Coefficient of determination: %.2f'% r2_score(y, y_pred))\n\nsns.regplot(x, y, ci=None).set_title(patient_id, fontsize=50);\npatients.loc[patients.Patient==patient_id,'gradient'] = regr.coef_[0]\npatients.loc[patients.Patient==patient_id]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_ids = patients['Patient']\n#patient_ids\n\nfor patient_id in patient_ids:\n    df_patient = pd.DataFrame(train.loc[train.Patient == patient_id])\n    x = np.array(train.loc[train.Patient == patient_id, 'Weeks'].values).reshape(-1,1)\n    y = train.loc[train.Patient == patient_id, 'FVC'].values\n    regr = linear_model.LinearRegression()\n    regr.fit(x,y)\n    patients.loc[patients.Patient==patient_id,'gradient'] = regr.coef_[0]\n    patients.loc[patients.Patient==patient_id,'intercept'] = regr.intercept_\n    patients.loc[patients.Patient==patient_id,'intercept'] = regr.intercept_\n    y_pred = regr.predict(x)\n    patients.loc[patients.Patient==patient_id,'rmse'] = np.sqrt(mean_squared_error(y, y_pred))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.title(\"Gradients\")\nmean = patients['gradient'].mean()\nstd = patients['gradient'].std()\nsns.distplot(patients['gradient'], bins=100);\nprint(f'Gradient mean: {mean}')\nprint(f'Gradient std: {std}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.title(\"Intercepts\")\nmean = patients['intercept'].mean()\nstd = patients['intercept'].std()\nsns.distplot(patients['intercept'], bins=100);\nprint(f'Intercept mean: {mean}')\nprint(f'Intercept std: {std}')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.title(\"RMSEs\")\nmean = patients['rmse'].mean()\nstd = patients['rmse'].std()\nsns.distplot(patients['rmse'], bins=100);\nprint(f'RMSE mean: {mean}')\nprint(f'RMSE std: {std}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr_matrix = patients.corr()\n#sns.heatmap(corr_matrix)\ncorr_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr_matrix.sort_values(by=['gradient'])[corr_matrix['gradient'].abs()>0.1]['gradient']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr_matrix.sort_values(by=['intercept'])[corr_matrix['intercept'].abs()>0.48]['intercept']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr_matrix.sort_values(by=['rmse'])[corr_matrix['rmse'].abs()>0.15]['rmse']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patients.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = patients['FVC_base']\ny = patients['gradient']\nfig, ax = plt.subplots(figsize=(10,10));\nax.set_title(\"gradient\")\nsns.scatterplot(x,np.log(y),hue=patients['Class'], size=patients['Age'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = patients['Class']\ny = patients['rmse']\nplt.title(\"rmse\")\nplt.scatter(x,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = patients['Class']\ny = patients['intercept']\nplt.title(\"intercept\")\nplt.scatter(x,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_features = ['Age', 'Class', 'Weeks_base', 'FVC_base',\n       'Percent_base', 'FVC_nom_base', 'gradient', 'intercept']\ntarget = ['rmse']\n\nx = patients[num_features]\ny = patients[target]\nregr = linear_model.LinearRegression()\nregr.fit(x,y)\nplt.scatter(num_features, regr.coef_)\nplt.xticks(rotation=90)\nplt.title(\"RMSE\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_features = ['Age', 'Class', 'Weeks_base', 'FVC_base',\n       'Percent_base', 'FVC_nom_base', 'gradient', 'rmse']\ntarget = ['intercept']\n\nx = patients[num_features]\ny = patients[target]\nregr = linear_model.LinearRegression()\nregr.fit(x,y)\nplt.scatter(num_features, regr.coef_)\nplt.xticks(rotation=90)\nplt.title(\"Intercept\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_features = ['Age', 'Class', 'Weeks_base', 'FVC_base',\n       'Percent_base', 'FVC_nom_base', 'intercept', 'rmse']\ntarget = ['gradient']\n\nx = patients[num_features]\ny = patients[target]\nregr = linear_model.LinearRegression()\nregr.fit(x,y)\nplt.scatter(num_features, regr.coef_)\nplt.xticks(rotation=90)\nplt.title(\"Gradient\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}