{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-30T00:15:51.995789Z","iopub.execute_input":"2022-06-30T00:15:51.996587Z","iopub.status.idle":"2022-06-30T00:15:52.010127Z","shell.execute_reply.started":"2022-06-30T00:15:51.996535Z","shell.execute_reply":"2022-06-30T00:15:52.008397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-family: Arials; font-size: 20px;text-align: center;; font-style: normal;line-height:1.3\">The objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile. </p>","metadata":{}},{"cell_type":"markdown","source":"The dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\n- `D_*` = Delinquency variables\n- `S_*` = Spend variables\n- `P_*` = Payment variables\n- `B_*` = Balance variables\n- `R_*` = Risk variables\n\nWith the following features being categorical:`B_30`,`B_38`,`D_114`,`D_116`,`D_117`,`D_120`,`D_126`,`D_63`,`D_64`, `D_66`,`D_68`\n\n\nYour task is to predict, for each customer_ID, the probability of a future payment default (target = 1).\n\nThanks to: https://www.kaggle.com/code/ripcurl/amex-eda-default-prediction/edit","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport plotly.express as px\nfrom itertools import cycle\n\nimport warnings, gc\nwarnings.filterwarnings('ignore')\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, roc_auc_score,confusion_matrix, classification_report\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:52.336577Z","iopub.execute_input":"2022-06-30T00:15:52.337031Z","iopub.status.idle":"2022-06-30T00:15:52.347421Z","shell.execute_reply.started":"2022-06-30T00:15:52.336998Z","shell.execute_reply":"2022-06-30T00:15:52.346306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr_lab = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ndf_tr_lab.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:52.845318Z","iopub.execute_input":"2022-06-30T00:15:52.845746Z","iopub.status.idle":"2022-06-30T00:15:53.579914Z","shell.execute_reply.started":"2022-06-30T00:15:52.845715Z","shell.execute_reply":"2022-06-30T00:15:53.578411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/amex-default-prediction/train_data.csv',\n                       nrows=30000)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:53.582905Z","iopub.execute_input":"2022-06-30T00:15:53.583453Z","iopub.status.idle":"2022-06-30T00:15:54.907465Z","shell.execute_reply.started":"2022-06-30T00:15:53.583406Z","shell.execute_reply":"2022-06-30T00:15:54.905896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:54.911003Z","iopub.execute_input":"2022-06-30T00:15:54.912481Z","iopub.status.idle":"2022-06-30T00:15:54.941304Z","shell.execute_reply.started":"2022-06-30T00:15:54.912437Z","shell.execute_reply":"2022-06-30T00:15:54.940342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.select_dtypes(include=np.object).head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:54.942909Z","iopub.execute_input":"2022-06-30T00:15:54.943541Z","iopub.status.idle":"2022-06-30T00:15:54.965274Z","shell.execute_reply.started":"2022-06-30T00:15:54.943496Z","shell.execute_reply":"2022-06-30T00:15:54.963859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.D_63.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:54.968395Z","iopub.execute_input":"2022-06-30T00:15:54.969277Z","iopub.status.idle":"2022-06-30T00:15:54.985430Z","shell.execute_reply.started":"2022-06-30T00:15:54.969227Z","shell.execute_reply":"2022-06-30T00:15:54.984436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df_train.D_63, df_tr_lab.iloc[:30000].target)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:54.986644Z","iopub.execute_input":"2022-06-30T00:15:54.987449Z","iopub.status.idle":"2022-06-30T00:15:55.018915Z","shell.execute_reply.started":"2022-06-30T00:15:54.987411Z","shell.execute_reply":"2022-06-30T00:15:55.017748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df_train.D_63, df_tr_lab.iloc[:30000].target).plot(kind='bar')\npd.crosstab(df_train.D_63, df_tr_lab.iloc[:30000].target).plot(kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:55.021757Z","iopub.execute_input":"2022-06-30T00:15:55.022361Z","iopub.status.idle":"2022-06-30T00:15:55.493779Z","shell.execute_reply.started":"2022-06-30T00:15:55.022321Z","shell.execute_reply":"2022-06-30T00:15:55.492152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## convert categorical variable to \"dummies\"\n* first fill null values","metadata":{}},{"cell_type":"code","source":"df_train['D_63'] = df_train['D_63'].fillna('CQ')","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:55.496623Z","iopub.execute_input":"2022-06-30T00:15:55.497156Z","iopub.status.idle":"2022-06-30T00:15:55.507582Z","shell.execute_reply.started":"2022-06-30T00:15:55.497105Z","shell.execute_reply":"2022-06-30T00:15:55.506725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.get_dummies(df_train, columns=['D_63'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:55.509275Z","iopub.execute_input":"2022-06-30T00:15:55.509851Z","iopub.status.idle":"2022-06-30T00:15:55.577342Z","shell.execute_reply.started":"2022-06-30T00:15:55.509816Z","shell.execute_reply":"2022-06-30T00:15:55.576101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.D_64.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:55.795538Z","iopub.execute_input":"2022-06-30T00:15:55.796651Z","iopub.status.idle":"2022-06-30T00:15:55.815103Z","shell.execute_reply.started":"2022-06-30T00:15:55.796591Z","shell.execute_reply":"2022-06-30T00:15:55.813742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* change -1 value to O (most common value)\n* fill null values with 'O' most common value","metadata":{}},{"cell_type":"code","source":"df_train['D_64'] = np.where(df_train['D_64']=='-1', 'O', df_train['D_64'])\ndf_train['D_64'] = df_train['D_64'].fillna('O')\ndf_train.D_64.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:56.059030Z","iopub.execute_input":"2022-06-30T00:15:56.059562Z","iopub.status.idle":"2022-06-30T00:15:56.089930Z","shell.execute_reply.started":"2022-06-30T00:15:56.059525Z","shell.execute_reply":"2022-06-30T00:15:56.088389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get dummies for categorical value","metadata":{}},{"cell_type":"code","source":"df_train = pd.get_dummies(df_train, columns=['D_64'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:56.322908Z","iopub.execute_input":"2022-06-30T00:15:56.323432Z","iopub.status.idle":"2022-06-30T00:15:56.438584Z","shell.execute_reply.started":"2022-06-30T00:15:56.323395Z","shell.execute_reply":"2022-06-30T00:15:56.436858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"drop for now the date column","metadata":{}},{"cell_type":"code","source":"df_train = df_train.drop(['S_2'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:56.592581Z","iopub.execute_input":"2022-06-30T00:15:56.593053Z","iopub.status.idle":"2022-06-30T00:15:56.638766Z","shell.execute_reply.started":"2022-06-30T00:15:56.593018Z","shell.execute_reply":"2022-06-30T00:15:56.637163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### fill missing values","metadata":{}},{"cell_type":"code","source":"df_mean = df_train.mean()\ndf_mean['P_2']","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:56.967920Z","iopub.execute_input":"2022-06-30T00:15:56.968370Z","iopub.status.idle":"2022-06-30T00:15:58.960714Z","shell.execute_reply.started":"2022-06-30T00:15:56.968335Z","shell.execute_reply":"2022-06-30T00:15:58.959276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:58.962525Z","iopub.execute_input":"2022-06-30T00:15:58.963320Z","iopub.status.idle":"2022-06-30T00:15:58.989336Z","shell.execute_reply.started":"2022-06-30T00:15:58.963283Z","shell.execute_reply":"2022-06-30T00:15:58.988133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for c in df_train.columns[1:]:\n    df_train[c] = df_train[c].fillna(df_mean[c])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:58.991700Z","iopub.execute_input":"2022-06-30T00:15:58.992149Z","iopub.status.idle":"2022-06-30T00:15:59.074857Z","shell.execute_reply.started":"2022-06-30T00:15:58.992114Z","shell.execute_reply":"2022-06-30T00:15:59.073279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## build a model","metadata":{}},{"cell_type":"code","source":"X_train = df_train.values[:, 1:]\nY_train = df_tr_lab['target'].values[0:30000]\nX_train.shape, Y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:59.077807Z","iopub.execute_input":"2022-06-30T00:15:59.078164Z","iopub.status.idle":"2022-06-30T00:15:59.868237Z","shell.execute_reply.started":"2022-06-30T00:15:59.078134Z","shell.execute_reply":"2022-06-30T00:15:59.866537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression\n\nlogreg = LogisticRegression(max_iter=300, solver='liblinear')\n\nlogreg.fit(X_train, Y_train)\n\nY_train_pred = logreg.predict(X_train)\n\n# score - Return the mean accuracy on the given test data and labels.\nlogreg.score(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:15:59.870655Z","iopub.execute_input":"2022-06-30T00:15:59.871166Z","iopub.status.idle":"2022-06-30T00:16:15.051027Z","shell.execute_reply.started":"2022-06-30T00:15:59.871118Z","shell.execute_reply":"2022-06-30T00:16:15.049564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('acc: ', accuracy_score(Y_train, Y_train_pred))\nconfusion_matrix(Y_train, Y_train_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:15.053076Z","iopub.execute_input":"2022-06-30T00:16:15.054558Z","iopub.status.idle":"2022-06-30T00:16:15.082544Z","shell.execute_reply.started":"2022-06-30T00:16:15.054500Z","shell.execute_reply":"2022-06-30T00:16:15.081158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Test the model\nas test we will use the next 10000","metadata":{}},{"cell_type":"code","source":"del df_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:15.084636Z","iopub.execute_input":"2022-06-30T00:16:15.085493Z","iopub.status.idle":"2022-06-30T00:16:16.502512Z","shell.execute_reply.started":"2022-06-30T00:16:15.085447Z","shell.execute_reply":"2022-06-30T00:16:16.500735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val = pd.read_csv('../input/amex-default-prediction/train_data.csv',\n                      nrows=30000, skiprows=range(1,-30000))\ndf_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:16.504680Z","iopub.execute_input":"2022-06-30T00:16:16.505084Z","iopub.status.idle":"2022-06-30T00:16:17.822098Z","shell.execute_reply.started":"2022-06-30T00:16:16.505016Z","shell.execute_reply":"2022-06-30T00:16:17.821012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:17.823430Z","iopub.execute_input":"2022-06-30T00:16:17.824396Z","iopub.status.idle":"2022-06-30T00:16:17.855922Z","shell.execute_reply.started":"2022-06-30T00:16:17.824354Z","shell.execute_reply":"2022-06-30T00:16:17.854340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### process validation data same pipeline as train","metadata":{}},{"cell_type":"code","source":"\ndf_val['D_63'] = df_val['D_63'].fillna('CQ')\ndf_val = pd.get_dummies(df_val, columns=['D_63'])\ndf_val['D_64'] = np.where(df_val['D_64']=='-1', 'O', df_val['D_64'])\ndf_val['D_64'] = df_val['D_64'].fillna('O')\ndf_val = pd.get_dummies(df_val, columns=['D_64'])\ndf_val = df_val.drop(['S_2'], axis=1)\nfor c in df_val.columns[1:]:\n    df_val[c] = df_val[c].fillna(df_mean[c])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:17.860434Z","iopub.execute_input":"2022-06-30T00:16:17.860897Z","iopub.status.idle":"2022-06-30T00:16:18.073447Z","shell.execute_reply.started":"2022-06-30T00:16:17.860853Z","shell.execute_reply":"2022-06-30T00:16:18.072370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_val = df_val.values[:, 1:]\nY_val = df_tr_lab['target'].values[-30000:]\nX_val.shape, Y_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:18.074940Z","iopub.execute_input":"2022-06-30T00:16:18.076426Z","iopub.status.idle":"2022-06-30T00:16:18.638733Z","shell.execute_reply.started":"2022-06-30T00:16:18.076370Z","shell.execute_reply":"2022-06-30T00:16:18.637351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = logreg.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:18.640463Z","iopub.execute_input":"2022-06-30T00:16:18.640887Z","iopub.status.idle":"2022-06-30T00:16:19.095394Z","shell.execute_reply.started":"2022-06-30T00:16:18.640852Z","shell.execute_reply":"2022-06-30T00:16:19.093867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_lg_prob = logreg.predict_proba(X_val)\nY_pred_lg_prob[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:19.097175Z","iopub.execute_input":"2022-06-30T00:16:19.097875Z","iopub.status.idle":"2022-06-30T00:16:19.596787Z","shell.execute_reply.started":"2022-06-30T00:16:19.097825Z","shell.execute_reply":"2022-06-30T00:16:19.595088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_score(Y_val, Y_pred_lg_prob[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:19.599024Z","iopub.execute_input":"2022-06-30T00:16:19.599729Z","iopub.status.idle":"2022-06-30T00:16:19.630704Z","shell.execute_reply.started":"2022-06-30T00:16:19.599678Z","shell.execute_reply":"2022-06-30T00:16:19.629174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"confusion_matrix(Y_val, Y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:19.632600Z","iopub.execute_input":"2022-06-30T00:16:19.633664Z","iopub.status.idle":"2022-06-30T00:16:19.653453Z","shell.execute_reply.started":"2022-06-30T00:16:19.633606Z","shell.execute_reply":"2022-06-30T00:16:19.652041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(Y_val, Y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:19.655626Z","iopub.execute_input":"2022-06-30T00:16:19.656546Z","iopub.status.idle":"2022-06-30T00:16:19.672105Z","shell.execute_reply.started":"2022-06-30T00:16:19.656494Z","shell.execute_reply":"2022-06-30T00:16:19.670669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(Y_val, Y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:19.674841Z","iopub.execute_input":"2022-06-30T00:16:19.675822Z","iopub.status.idle":"2022-06-30T00:16:19.748974Z","shell.execute_reply.started":"2022-06-30T00:16:19.675768Z","shell.execute_reply":"2022-06-30T00:16:19.747310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lets try and to better\n### KNN","metadata":{}},{"cell_type":"code","source":"knn15 = KNeighborsClassifier(15).fit(X_train, Y_train)\nY_train_pred = knn15.predict(X_train)\n\n# score - Return the mean accuracy on the given test data and labels.\nknn15.score(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:16:19.750743Z","iopub.execute_input":"2022-06-30T00:16:19.751322Z","iopub.status.idle":"2022-06-30T00:17:03.471792Z","shell.execute_reply.started":"2022-06-30T00:16:19.751266Z","shell.execute_reply":"2022-06-30T00:17:03.470253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('acc: ', accuracy_score(Y_train, Y_train_pred))\nconfusion_matrix(Y_train, Y_train_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:17:03.473644Z","iopub.execute_input":"2022-06-30T00:17:03.474015Z","iopub.status.idle":"2022-06-30T00:17:03.494656Z","shell.execute_reply.started":"2022-06-30T00:17:03.473984Z","shell.execute_reply":"2022-06-30T00:17:03.493494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = knn15.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:17:03.495836Z","iopub.execute_input":"2022-06-30T00:17:03.496966Z","iopub.status.idle":"2022-06-30T00:17:24.763016Z","shell.execute_reply.started":"2022-06-30T00:17:03.496914Z","shell.execute_reply":"2022-06-30T00:17:24.761939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(confusion_matrix(Y_val, Y_pred))\nprint(accuracy_score(Y_val, Y_pred))\nprint(classification_report(Y_val, Y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:17:24.765698Z","iopub.execute_input":"2022-06-30T00:17:24.766236Z","iopub.status.idle":"2022-06-30T00:17:24.833034Z","shell.execute_reply.started":"2022-06-30T00:17:24.766185Z","shell.execute_reply":"2022-06-30T00:17:24.831136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_knn_prob = knn15.predict_proba(X_val)\nY_pred_knn_prob[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:17:24.834877Z","iopub.execute_input":"2022-06-30T00:17:24.835851Z","iopub.status.idle":"2022-06-30T00:17:45.993945Z","shell.execute_reply.started":"2022-06-30T00:17:24.835784Z","shell.execute_reply":"2022-06-30T00:17:45.992581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_score(Y_val, Y_pred_knn_prob[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:17:45.996129Z","iopub.execute_input":"2022-06-30T00:17:45.996483Z","iopub.status.idle":"2022-06-30T00:17:46.016875Z","shell.execute_reply.started":"2022-06-30T00:17:45.996452Z","shell.execute_reply":"2022-06-30T00:17:46.015920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## And the king: RandomForest\n* next week we will learn about it","metadata":{}},{"cell_type":"code","source":"rf = RandomForestClassifier(6).fit(X_train, Y_train)\nY_train_pred = knn15.predict(X_train)\n\n# score - Return the mean accuracy on the given test data and labels.\nprint(rf.score(X_train, Y_train))\nprint(confusion_matrix( Y_train, Y_train_pred))\nprint(accuracy_score( Y_train, Y_train_pred))\nprint(classification_report( Y_train, Y_train_pred))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:17:46.018008Z","iopub.execute_input":"2022-06-30T00:17:46.018968Z","iopub.status.idle":"2022-06-30T00:18:14.385294Z","shell.execute_reply.started":"2022-06-30T00:17:46.018934Z","shell.execute_reply":"2022-06-30T00:18:14.383474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = rf.predict(X_val)\nprint('Test confusion matrix:\\n',confusion_matrix(Y_val, Y_pred))\nprint('Test acc: ',accuracy_score(Y_val, Y_pred))\nprint(classification_report(Y_val, Y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:14.387352Z","iopub.execute_input":"2022-06-30T00:18:14.387747Z","iopub.status.idle":"2022-06-30T00:18:14.958264Z","shell.execute_reply.started":"2022-06-30T00:18:14.387714Z","shell.execute_reply":"2022-06-30T00:18:14.956943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_rf_prob_train = rf.predict_proba(X_train)\nY_pred_rf_prob_train[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:14.959711Z","iopub.execute_input":"2022-06-30T00:18:14.960561Z","iopub.status.idle":"2022-06-30T00:18:15.491665Z","shell.execute_reply.started":"2022-06-30T00:18:14.960518Z","shell.execute_reply":"2022-06-30T00:18:15.490251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'auc train: ', roc_auc_score(Y_train, Y_pred_rf_prob_train[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:15.498840Z","iopub.execute_input":"2022-06-30T00:18:15.499748Z","iopub.status.idle":"2022-06-30T00:18:15.520689Z","shell.execute_reply.started":"2022-06-30T00:18:15.499702Z","shell.execute_reply":"2022-06-30T00:18:15.519230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_rf_prob = rf.predict_proba(X_val)\nY_pred_rf_prob[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:15.522448Z","iopub.execute_input":"2022-06-30T00:18:15.522866Z","iopub.status.idle":"2022-06-30T00:18:16.060812Z","shell.execute_reply.started":"2022-06-30T00:18:15.522831Z","shell.execute_reply":"2022-06-30T00:18:16.059312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_score(Y_train, Y_pred_rf_prob[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.063167Z","iopub.execute_input":"2022-06-30T00:18:16.063619Z","iopub.status.idle":"2022-06-30T00:18:16.085843Z","shell.execute_reply.started":"2022-06-30T00:18:16.063586Z","shell.execute_reply":"2022-06-30T00:18:16.084047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ensemble the models","metadata":{}},{"cell_type":"code","source":"Y_pred_ensemble_p = (Y_pred_lg_prob+Y_pred_knn_prob+Y_pred_rf_prob)/3.","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.088753Z","iopub.execute_input":"2022-06-30T00:18:16.089390Z","iopub.status.idle":"2022-06-30T00:18:16.097126Z","shell.execute_reply.started":"2022-06-30T00:18:16.089333Z","shell.execute_reply":"2022-06-30T00:18:16.095280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_ensemble = np.argmax(Y_pred_ensemble_p, axis=1)\nY_pred_ensemble[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.099384Z","iopub.execute_input":"2022-06-30T00:18:16.099825Z","iopub.status.idle":"2022-06-30T00:18:16.113526Z","shell.execute_reply.started":"2022-06-30T00:18:16.099793Z","shell.execute_reply":"2022-06-30T00:18:16.112344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'auc test:', roc_auc_score(Y_val, Y_pred_ensemble_p[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.115615Z","iopub.execute_input":"2022-06-30T00:18:16.116520Z","iopub.status.idle":"2022-06-30T00:18:16.140337Z","shell.execute_reply.started":"2022-06-30T00:18:16.116480Z","shell.execute_reply":"2022-06-30T00:18:16.138924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(confusion_matrix(Y_val, Y_pred_ensemble))\nprint('test acc: ',accuracy_score(Y_val, Y_pred_ensemble))\nprint(classification_report(Y_val, Y_pred_ensemble))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.142267Z","iopub.execute_input":"2022-06-30T00:18:16.142667Z","iopub.status.idle":"2022-06-30T00:18:16.220586Z","shell.execute_reply.started":"2022-06-30T00:18:16.142633Z","shell.execute_reply":"2022-06-30T00:18:16.218621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets resample the data","metadata":{}},{"cell_type":"code","source":"def preprocess(df):\n    df['D_63'] = df['D_63'].fillna('CQ')\n    df = pd.get_dummies(df, columns=['D_63'])\n    df['D_64'] = np.where(df['D_64']=='-1', 'O', df['D_64'])\n    df['D_64'] = df['D_64'].fillna('O')\n    df = pd.get_dummies(df, columns=['D_64'])\n    df = df.drop(['S_2'], axis=1)\n    \n    for c in df.columns[1:]:\n        df[c] = df[c].fillna(df_mean[c])\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.224340Z","iopub.execute_input":"2022-06-30T00:18:16.225026Z","iopub.status.idle":"2022-06-30T00:18:16.236576Z","shell.execute_reply.started":"2022-06-30T00:18:16.224944Z","shell.execute_reply":"2022-06-30T00:18:16.235030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr_lab.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.239388Z","iopub.execute_input":"2022-06-30T00:18:16.240006Z","iopub.status.idle":"2022-06-30T00:18:16.259043Z","shell.execute_reply.started":"2022-06-30T00:18:16.239952Z","shell.execute_reply":"2022-06-30T00:18:16.257576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/amex-default-prediction/train_data.csv',\n                       nrows=30000)\ndf_train['target'] = df_tr_lab['target'].values[:30000]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:16.261563Z","iopub.execute_input":"2022-06-30T00:18:16.262560Z","iopub.status.idle":"2022-06-30T00:18:17.558146Z","shell.execute_reply.started":"2022-06-30T00:18:16.262431Z","shell.execute_reply":"2022-06-30T00:18:17.556798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k=0\ndf_tmp = pd.read_csv('../input/amex-default-prediction/train_data.csv',\n                       nrows=10000, skiprows=(1, 30000+k*10000))\ndf_tmp['target'] = df_tr_lab['target'].values[30000+(k)*10000:30000+(k+1)*10000]\ndf_train = pd.concat([df_train, df_tmp], axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:17.562673Z","iopub.execute_input":"2022-06-30T00:18:17.563101Z","iopub.status.idle":"2022-06-30T00:18:18.015700Z","shell.execute_reply.started":"2022-06-30T00:18:17.563053Z","shell.execute_reply":"2022-06-30T00:18:18.014675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:18.017410Z","iopub.execute_input":"2022-06-30T00:18:18.017788Z","iopub.status.idle":"2022-06-30T00:18:18.024842Z","shell.execute_reply.started":"2022-06-30T00:18:18.017754Z","shell.execute_reply":"2022-06-30T00:18:18.023933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor k in range(5):\n    df_tmp = pd.read_csv('../input/amex-default-prediction/train_data.csv',\n                       nrows=10000, skiprows=(1, 30000+k*10000))\n    df_tmp['target'] = df_tr_lab['target'].values[30000+(k)*10000:30000+(k+1)*10000]\n    \n    df_train = pd.concat([df_train, df_tmp[df_tmp.target==1]], axis=0)\n    print(df_train.shape, df_train.target.sum())                    ","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:18.026205Z","iopub.execute_input":"2022-06-30T00:18:18.026836Z","iopub.status.idle":"2022-06-30T00:18:20.311452Z","shell.execute_reply.started":"2022-06-30T00:18:18.026803Z","shell.execute_reply":"2022-06-30T00:18:20.310012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape, df_train.target.value_counts(),df_train[df_train.target==1].sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:20.313640Z","iopub.execute_input":"2022-06-30T00:18:20.314659Z","iopub.status.idle":"2022-06-30T00:18:21.637813Z","shell.execute_reply.started":"2022-06-30T00:18:20.314615Z","shell.execute_reply":"2022-06-30T00:18:21.636004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:18:21.639879Z","iopub.execute_input":"2022-06-30T00:18:21.640475Z","iopub.status.idle":"2022-06-30T00:18:21.654102Z","shell.execute_reply.started":"2022-06-30T00:18:21.640431Z","shell.execute_reply":"2022-06-30T00:18:21.652419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_train = pd.concat([preprocess(df_train.drop('target', axis=1)),\n                      df_train['target']], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:21:14.922876Z","iopub.execute_input":"2022-06-30T00:21:14.923423Z","iopub.status.idle":"2022-06-30T00:21:15.598833Z","shell.execute_reply.started":"2022-06-30T00:21:14.923381Z","shell.execute_reply":"2022-06-30T00:21:15.597638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df_train.values[:, 1:-1]\nY_train = df_train['target'].values\nX_train.shape, Y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:21:43.194759Z","iopub.execute_input":"2022-06-30T00:21:43.195213Z","iopub.status.idle":"2022-06-30T00:21:44.342250Z","shell.execute_reply.started":"2022-06-30T00:21:43.195178Z","shell.execute_reply":"2022-06-30T00:21:44.340754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# logistic regression\n\nlogreg = LogisticRegression(max_iter=1000, solver='liblinear')\n\nlogreg.fit(X_train, Y_train)\n\nY_train_pred = logreg.predict(X_train)\n\n# score - Return the mean accuracy on the given test data and labels.\nlogreg.score(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:33:00.273820Z","iopub.execute_input":"2022-06-30T00:33:00.274293Z","iopub.status.idle":"2022-06-30T00:33:44.828990Z","shell.execute_reply.started":"2022-06-30T00:33:00.274260Z","shell.execute_reply":"2022-06-30T00:33:44.827634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train acc: ', accuracy_score(Y_train, Y_train_pred))\nconfusion_matrix(Y_train, Y_train_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:32:21.188843Z","iopub.execute_input":"2022-06-30T00:32:21.189572Z","iopub.status.idle":"2022-06-30T00:32:21.218015Z","shell.execute_reply.started":"2022-06-30T00:32:21.189535Z","shell.execute_reply":"2022-06-30T00:32:21.216582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = logreg.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:32:24.180524Z","iopub.execute_input":"2022-06-30T00:32:24.180994Z","iopub.status.idle":"2022-06-30T00:32:24.688691Z","shell.execute_reply.started":"2022-06-30T00:32:24.180959Z","shell.execute_reply":"2022-06-30T00:32:24.687468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('test acc: ', accuracy_score(Y_val, Y_pred))\nconfusion_matrix(Y_val, Y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:32:26.905396Z","iopub.execute_input":"2022-06-30T00:32:26.906276Z","iopub.status.idle":"2022-06-30T00:32:26.924820Z","shell.execute_reply.started":"2022-06-30T00:32:26.906237Z","shell.execute_reply":"2022-06-30T00:32:26.923509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_lg_prob = logreg.predict_proba(X_val)\nY_pred_knn_prob[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:32:30.903677Z","iopub.execute_input":"2022-06-30T00:32:30.904409Z","iopub.status.idle":"2022-06-30T00:32:31.394420Z","shell.execute_reply.started":"2022-06-30T00:32:30.904370Z","shell.execute_reply":"2022-06-30T00:32:31.393141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_score(Y_val, Y_pred_lg_prob[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:32:33.605360Z","iopub.execute_input":"2022-06-30T00:32:33.605837Z","iopub.status.idle":"2022-06-30T00:32:33.624046Z","shell.execute_reply.started":"2022-06-30T00:32:33.605800Z","shell.execute_reply":"2022-06-30T00:32:33.623166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict for test and submit","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\ndf_subm.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:42:46.762178Z","iopub.execute_input":"2022-06-30T00:42:46.762654Z","iopub.status.idle":"2022-06-30T00:42:48.180321Z","shell.execute_reply.started":"2022-06-30T00:42:46.762622Z","shell.execute_reply":"2022-06-30T00:42:48.179079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_subm)//30000","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:47:35.908858Z","iopub.execute_input":"2022-06-30T00:47:35.909305Z","iopub.status.idle":"2022-06-30T00:47:35.917692Z","shell.execute_reply.started":"2022-06-30T00:47:35.909272Z","shell.execute_reply":"2022-06-30T00:47:35.916648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = []\nind = 0\nfor k in range(31):\n    start_idx = k*30000\n    end_idx = min((k+1)*30000, len(df_subm))\n    df_test = pd.read_csv('../input/amex-default-prediction/test_data.csv',\n                          nrows=end_idx-start_idx, \n                          skiprows=(1+k*30000, end_idx))\n    df_test = preprocess(df_test)\n    print(df_test.shape)\n    X = df_test.values[:, 1:]\n    pred_n = logreg.predict_proba(X)[:,1]\n    print(pred_n.shape)\n    pred += list(pred_n)            \n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:49:13.227566Z","iopub.execute_input":"2022-06-30T00:49:13.228136Z","iopub.status.idle":"2022-06-30T00:50:38.948472Z","shell.execute_reply.started":"2022-06-30T00:49:13.228054Z","shell.execute_reply":"2022-06-30T00:50:38.946325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:53:30.261799Z","iopub.execute_input":"2022-06-30T00:53:30.262433Z","iopub.status.idle":"2022-06-30T00:53:30.270892Z","shell.execute_reply.started":"2022-06-30T00:53:30.262382Z","shell.execute_reply":"2022-06-30T00:53:30.269591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm['prediction'] = pred","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:53:33.800667Z","iopub.execute_input":"2022-06-30T00:53:33.801131Z","iopub.status.idle":"2022-06-30T00:53:34.119290Z","shell.execute_reply.started":"2022-06-30T00:53:33.801095Z","shell.execute_reply":"2022-06-30T00:53:34.118000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:58:43.150969Z","iopub.execute_input":"2022-06-30T00:58:43.151604Z","iopub.status.idle":"2022-06-30T00:58:43.169667Z","shell.execute_reply.started":"2022-06-30T00:58:43.151554Z","shell.execute_reply":"2022-06-30T00:58:43.168633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-30T01:08:28.141729Z","iopub.execute_input":"2022-06-30T01:08:28.142318Z","iopub.status.idle":"2022-06-30T01:08:28.150513Z","shell.execute_reply.started":"2022-06-30T01:08:28.142279Z","shell.execute_reply":"2022-06-30T01:08:28.149655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T00:59:12.825850Z","iopub.execute_input":"2022-06-30T00:59:12.827139Z","iopub.status.idle":"2022-06-30T00:59:18.287752Z","shell.execute_reply.started":"2022-06-30T00:59:12.827029Z","shell.execute_reply":"2022-06-30T00:59:18.286411Z"},"trusted":true},"execution_count":null,"outputs":[]}]}