{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nImportance Of Problem Statement\n\n•Survival Prediction after hospitalization is important for doctors and the patient or their family members.\n\n• It helps doctors to understand the degree of severity and accordingly plan the medical treatment.\n\n• Helps the medical staff to prioritize patients based on fatality of medical condition.\n\n• Patients and their families can get adequate time to make necessary arrangements.\n\n• Leads to timely prevention and treatment and worse treatment decisions (over treatment or late palliative care) can be avoided.\n\nThere are many studies regarding survival prediction for certain specific diseases like cancer, sepsis, brain tumor, but very few studies which include many diseases together.\n\nWith your model you can bridge this gap by including as many diseases which can adversely affect a patient’s survival. These diseases are namely liver cirrhosis, cardiovascular, respiratory failure, trauma, sepsis, metabolic, neurologic and gastrointestinal diseases.\n\nAs part of this notebook we would look at basic EDA and a baseline model to predict survival rate of a patient.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.impute import KNNImputer\nplt.style.use('fivethirtyeight')\npd.set_option('max_columns', 500)\npd.set_option('max_rows',100)#for displaying all columns\ncolor_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-10T06:11:09.140854Z","iopub.execute_input":"2022-07-10T06:11:09.141334Z","iopub.status.idle":"2022-07-10T06:11:09.15692Z","shell.execute_reply.started":"2022-07-10T06:11:09.141296Z","shell.execute_reply":"2022-07-10T06:11:09.155987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/patient-survival-prediction/train.csv\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-10T04:21:27.884883Z","iopub.execute_input":"2022-07-10T04:21:27.885298Z","iopub.status.idle":"2022-07-10T04:21:29.208239Z","shell.execute_reply.started":"2022-07-10T04:21:27.885251Z","shell.execute_reply":"2022-07-10T04:21:29.207115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T04:35:35.512943Z","iopub.execute_input":"2022-07-10T04:35:35.513533Z","iopub.status.idle":"2022-07-10T04:35:35.616868Z","shell.execute_reply.started":"2022-07-10T04:35:35.513482Z","shell.execute_reply":"2022-07-10T04:35:35.615685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T04:32:32.344353Z","iopub.execute_input":"2022-07-10T04:32:32.344783Z","iopub.status.idle":"2022-07-10T04:32:32.423731Z","shell.execute_reply.started":"2022-07-10T04:32:32.344746Z","shell.execute_reply":"2022-07-10T04:32:32.422512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T04:46:56.833739Z","iopub.execute_input":"2022-07-10T04:46:56.83413Z","iopub.status.idle":"2022-07-10T04:46:56.952874Z","shell.execute_reply.started":"2022-07-10T04:46:56.8341Z","shell.execute_reply":"2022-07-10T04:46:56.951567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-10T04:47:19.146955Z","iopub.execute_input":"2022-07-10T04:47:19.147387Z","iopub.status.idle":"2022-07-10T04:47:19.156709Z","shell.execute_reply.started":"2022-07-10T04:47:19.147352Z","shell.execute_reply":"2022-07-10T04:47:19.155294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_missing = pd.DataFrame([train.isna().mean()]).T\ncount_missing = count_missing.rename(columns={0: \"train_missing\"})\n\ncount_missing.query(\"train_missing > 0\").plot(kind=\"barh\", figsize=(15, 30), title=\"% of Values Missing\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T06:09:23.015805Z","iopub.execute_input":"2022-07-10T06:09:23.017015Z","iopub.status.idle":"2022-07-10T06:09:24.326783Z","shell.execute_reply.started":"2022-07-10T06:09:23.016972Z","shell.execute_reply":"2022-07-10T06:09:24.325726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations\n\n* Unnamed:83 has only NA values, we can drop it.\n\n* We would use KNN imputer for the rest of the data.","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['elective_surgery', 'ethnicity', 'gender','icu_stay_type', \\\n                       'icu_type','icu_admit_source', 'apache_post_operative', \\\n                       'arf_apache', 'gcs_eyes_apache', 'gcs_motor_apache', \\\n                       'gcs_unable_apache', 'gcs_verbal_apache', \\\n                       'intubated_apache', 'ventilated_apache', 'aids', 'cirrhosis', 'diabetes_mellitus', \\\n                       'hepatic_failure', 'immunosuppression', 'leukemia', 'lymphoma', \\\n                       'solid_tumor_with_metastasis', 'apache_3j_bodysystem', \\\n                       'apache_2_bodysystem']\ni = 1\nplt.figure(figsize=(15,35))\n\nfor columns in categorical_columns:\n    plt.subplot(6,4,i)\n    sns.countplot(x=columns,hue=\"hospital_death\",data=train)\n    plt.xticks(rotation=45)\n    plt.yticks([])\n    i+=1\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T05:12:35.162505Z","iopub.execute_input":"2022-07-10T05:12:35.162929Z","iopub.status.idle":"2022-07-10T05:12:38.483052Z","shell.execute_reply.started":"2022-07-10T05:12:35.162898Z","shell.execute_reply":"2022-07-10T05:12:38.481689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Observations:\n\n* Patients with gcs_verbal_apache score of 1 have the most number of deaths even though the proportion of people with a score of 1 is much lesser than people with score of 6 on the same test.\n\n* Major Cause of death according to the apache_3j_bodysystem test is CardioVascular followed by Sepsis","metadata":{}},{"cell_type":"code","source":"non_categorical_columns = ['hospital_id', 'age', 'bmi','height',\\\n       'icu_id', 'pre_icu_los_days', 'weight',\\\n       'apache_2_diagnosis', 'apache_3j_diagnosis','heart_rate_apache',\\\n        'map_apache', 'resprate_apache', 'temp_apache',\\\n       'ventilated_apache', 'd1_diasbp_max', 'd1_diasbp_min',\\\n       'd1_diasbp_noninvasive_max', 'd1_diasbp_noninvasive_min',\\\n       'd1_heartrate_max', 'd1_heartrate_min', 'd1_mbp_max', 'd1_mbp_min',\\\n       'd1_mbp_noninvasive_max', 'd1_mbp_noninvasive_min', 'd1_resprate_max',\\\n       'd1_resprate_min', 'd1_spo2_max', 'd1_spo2_min', 'd1_sysbp_max',\\\n       'd1_sysbp_min', 'd1_sysbp_noninvasive_max', 'd1_sysbp_noninvasive_min',\\\n       'd1_temp_max', 'd1_temp_min', 'h1_diasbp_max', 'h1_diasbp_min',\\\n       'h1_diasbp_noninvasive_max', 'h1_diasbp_noninvasive_min',\\\n       'h1_heartrate_max', 'h1_heartrate_min', 'h1_mbp_max', 'h1_mbp_min',\\\n       'h1_mbp_noninvasive_max', 'h1_mbp_noninvasive_min', 'h1_resprate_max',\\\n       'h1_resprate_min', 'h1_spo2_max', 'h1_spo2_min', 'h1_sysbp_max',\\\n       'h1_sysbp_min', 'h1_sysbp_noninvasive_max', 'h1_sysbp_noninvasive_min',\\\n       'd1_glucose_max', 'd1_glucose_min', 'd1_potassium_max',\\\n       'd1_potassium_min', 'apache_4a_hospital_death_prob',\\\n       'apache_4a_icu_death_prob']\ni = 1\nplt.figure(figsize=(20,20))\nfor columns in non_categorical_columns:\n    plt.subplot(15,4,i)\n    sns.kdeplot(x=columns,data=train)\n    #plt.title(columns)\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2022-07-10T05:47:44.217133Z","iopub.execute_input":"2022-07-10T05:47:44.217594Z","iopub.status.idle":"2022-07-10T05:48:06.231896Z","shell.execute_reply.started":"2022-07-10T05:47:44.217555Z","shell.execute_reply":"2022-07-10T05:48:06.230535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess the data","metadata":{}},{"cell_type":"code","source":"le = preprocessing.OrdinalEncoder()\ntrain[categorical_columns] = le.fit_transform(train[categorical_columns])\ny = train['hospital_death']\nX = train.drop(['hospital_death','encounter_id', 'patient_id','Unnamed: 83'],axis=1)\nimputer = KNNImputer(n_neighbors=10)\nX=imputer.fit_transform(X)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\nscaler = preprocessing.StandardScaler().fit(X_train)\nscaled = scaler.transform(X_train)\nscaled_test = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T06:45:59.448605Z","iopub.execute_input":"2022-07-10T06:45:59.449024Z","iopub.status.idle":"2022-07-10T07:02:37.5138Z","shell.execute_reply.started":"2022-07-10T06:45:59.448992Z","shell.execute_reply":"2022-07-10T07:02:37.512528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = RandomForestClassifier(class_weight=\"balanced\")\nclf.fit(scaled,y_train)\npred = clf.predict(scaled_test)\nroc_auc_score(y_test,pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:26:30.432101Z","iopub.execute_input":"2022-07-10T07:26:30.432546Z","iopub.status.idle":"2022-07-10T07:26:48.510032Z","shell.execute_reply.started":"2022-07-10T07:26:30.43251Z","shell.execute_reply":"2022-07-10T07:26:48.508632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST = pd.read_csv(\"../input/patient-survival-prediction/test.csv\")\nTEST[categorical_columns] = le.fit_transform(TEST[categorical_columns])\nX = TEST.drop(['encounter_id', 'patient_id','Unnamed: 83'],axis=1)\nX=imputer.transform(X)\nscaled = scaler.transform(X)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:57:39.073595Z","iopub.execute_input":"2022-07-10T07:57:39.074724Z","iopub.status.idle":"2022-07-10T08:04:47.525549Z","shell.execute_reply.started":"2022-07-10T07:57:39.074665Z","shell.execute_reply":"2022-07-10T08:04:47.524079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = clf.predict_proba(scaled)\nprediction = pd.DataFrame()\nprediction['patient_id'] = TEST['patient_id']\nprediction['hospital_death'] = pred[:,1]\nprediction.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:08:02.210892Z","iopub.execute_input":"2022-07-10T08:08:02.211372Z","iopub.status.idle":"2022-07-10T08:08:03.077034Z","shell.execute_reply.started":"2022-07-10T08:08:02.211336Z","shell.execute_reply":"2022-07-10T08:08:03.07577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-10T08:12:31.377544Z","iopub.execute_input":"2022-07-10T08:12:31.377994Z","iopub.status.idle":"2022-07-10T08:12:39.332145Z","shell.execute_reply.started":"2022-07-10T08:12:31.377957Z","shell.execute_reply":"2022-07-10T08:12:39.330829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}