{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Importing Libraries**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom keras.callbacks import EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:42.048374Z","iopub.execute_input":"2022-11-05T12:42:42.048791Z","iopub.status.idle":"2022-11-05T12:42:42.058476Z","shell.execute_reply.started":"2022-11-05T12:42:42.048757Z","shell.execute_reply":"2022-11-05T12:42:42.056668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the Dataset\n\n\nLoading the 2 most recent transactions of the customers.","metadata":{}},{"cell_type":"code","source":"number_of_transactions = 2","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:42.061596Z","iopub.execute_input":"2022-11-05T12:42:42.062167Z","iopub.status.idle":"2022-11-05T12:42:42.070293Z","shell.execute_reply.started":"2022-11-05T12:42:42.062118Z","shell.execute_reply":"2022-11-05T12:42:42.069330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=200000).groupby('customer_ID').tail(number_of_transactions).set_index('customer_ID', drop=True).sort_index()\nlabels = pd.read_csv('../input/amex-default-prediction/train_labels.csv').set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:42.072114Z","iopub.execute_input":"2022-11-05T12:42:42.072520Z","iopub.status.idle":"2022-11-05T12:42:49.923481Z","shell.execute_reply.started":"2022-11-05T12:42:42.072484Z","shell.execute_reply":"2022-11-05T12:42:49.922328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Merging the Training data with Labels**","metadata":{}},{"cell_type":"code","source":"train_df = pd.merge(data, labels, left_index=True, right_index=True)  ","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:49.925111Z","iopub.execute_input":"2022-11-05T12:42:49.925471Z","iopub.status.idle":"2022-11-05T12:42:50.019473Z","shell.execute_reply.started":"2022-11-05T12:42:49.925439Z","shell.execute_reply":"2022-11-05T12:42:50.018427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:50.022783Z","iopub.execute_input":"2022-11-05T12:42:50.023119Z","iopub.status.idle":"2022-11-05T12:42:50.052051Z","shell.execute_reply.started":"2022-11-05T12:42:50.023090Z","shell.execute_reply":"2022-11-05T12:42:50.050730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The number of observations in Training Dataset is :\",len(train_df))","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:50.053302Z","iopub.execute_input":"2022-11-05T12:42:50.053669Z","iopub.status.idle":"2022-11-05T12:42:50.059020Z","shell.execute_reply.started":"2022-11-05T12:42:50.053637Z","shell.execute_reply":"2022-11-05T12:42:50.058278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:50.060820Z","iopub.execute_input":"2022-11-05T12:42:50.061152Z","iopub.status.idle":"2022-11-05T12:42:50.092601Z","shell.execute_reply.started":"2022-11-05T12:42:50.061120Z","shell.execute_reply":"2022-11-05T12:42:50.091310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Descriptive Statistics**","metadata":{}},{"cell_type":"code","source":"train_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:50.094303Z","iopub.execute_input":"2022-11-05T12:42:50.094751Z","iopub.status.idle":"2022-11-05T12:42:50.820188Z","shell.execute_reply.started":"2022-11-05T12:42:50.094707Z","shell.execute_reply":"2022-11-05T12:42:50.818957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Visualisation\n\n**Count Plot of the Target Variable**","metadata":{}},{"cell_type":"code","source":"sns.countplot(x = 'target',data = train_df)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:50.821855Z","iopub.execute_input":"2022-11-05T12:42:50.822346Z","iopub.status.idle":"2022-11-05T12:42:51.023669Z","shell.execute_reply.started":"2022-11-05T12:42:50.822300Z","shell.execute_reply":"2022-11-05T12:42:51.022466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Processing","metadata":{}},{"cell_type":"markdown","source":"**Dropping the Transaction Dates**","metadata":{}},{"cell_type":"code","source":"drop_cols = ['S_2'] \ntrain_df.drop(drop_cols, inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.025355Z","iopub.execute_input":"2022-11-05T12:42:51.026693Z","iopub.status.idle":"2022-11-05T12:42:51.046367Z","shell.execute_reply.started":"2022-11-05T12:42:51.026641Z","shell.execute_reply":"2022-11-05T12:42:51.044991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating Training Labels and Data**","metadata":{}},{"cell_type":"code","source":"train_df1 = pd.get_dummies(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.047785Z","iopub.execute_input":"2022-11-05T12:42:51.048427Z","iopub.status.idle":"2022-11-05T12:42:51.109156Z","shell.execute_reply.started":"2022-11-05T12:42:51.048374Z","shell.execute_reply":"2022-11-05T12:42:51.107667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_df1['target']\nX = train_df1.drop('target', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.110959Z","iopub.execute_input":"2022-11-05T12:42:51.111561Z","iopub.status.idle":"2022-11-05T12:42:51.159206Z","shell.execute_reply.started":"2022-11-05T12:42:51.111514Z","shell.execute_reply":"2022-11-05T12:42:51.157780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Handling missing values**\n\nMissing values are imputed with the respective column mean.","metadata":{}},{"cell_type":"code","source":"col_names = X.columns\nimputer = SimpleImputer()\nX = pd.DataFrame(imputer.fit_transform(X))  \nX.columns = col_names","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.160873Z","iopub.execute_input":"2022-11-05T12:42:51.161814Z","iopub.status.idle":"2022-11-05T12:42:51.363199Z","shell.execute_reply.started":"2022-11-05T12:42:51.161776Z","shell.execute_reply":"2022-11-05T12:42:51.361913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Standardization**","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nX = pd.DataFrame(scaler.fit_transform(X), index=X.index, columns=X.columns) ","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.368912Z","iopub.execute_input":"2022-11-05T12:42:51.369285Z","iopub.status.idle":"2022-11-05T12:42:51.456600Z","shell.execute_reply.started":"2022-11-05T12:42:51.369254Z","shell.execute_reply":"2022-11-05T12:42:51.455300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Splitting into training and validation sets**","metadata":{}},{"cell_type":"code","source":"train_X, val_X, train_y, val_y = train_test_split(X, y, random_state = 1, shuffle=True, stratify=y, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.458577Z","iopub.execute_input":"2022-11-05T12:42:51.459032Z","iopub.status.idle":"2022-11-05T12:42:51.527598Z","shell.execute_reply.started":"2022-11-05T12:42:51.458989Z","shell.execute_reply":"2022-11-05T12:42:51.526721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Memory cleanup to prevent out of memory error.","metadata":{}},{"cell_type":"code","source":"del data, labels, X, y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.528745Z","iopub.execute_input":"2022-11-05T12:42:51.529623Z","iopub.status.idle":"2022-11-05T12:42:51.767295Z","shell.execute_reply.started":"2022-11-05T12:42:51.529582Z","shell.execute_reply":"2022-11-05T12:42:51.766005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building","metadata":{}},{"cell_type":"markdown","source":"**Logistic Regression**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report,accuracy_score\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.769216Z","iopub.execute_input":"2022-11-05T12:42:51.769756Z","iopub.status.idle":"2022-11-05T12:42:51.777530Z","shell.execute_reply.started":"2022-11-05T12:42:51.769705Z","shell.execute_reply":"2022-11-05T12:42:51.776301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.779316Z","iopub.execute_input":"2022-11-05T12:42:51.779818Z","iopub.status.idle":"2022-11-05T12:42:51.791459Z","shell.execute_reply.started":"2022-11-05T12:42:51.779768Z","shell.execute_reply":"2022-11-05T12:42:51.790046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_lr = LogisticRegression(n_jobs=1, C=1e5)\nclf_lr.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:51.793179Z","iopub.execute_input":"2022-11-05T12:42:51.793701Z","iopub.status.idle":"2022-11-05T12:42:52.783380Z","shell.execute_reply.started":"2022-11-05T12:42:51.793654Z","shell.execute_reply":"2022-11-05T12:42:52.782198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ny_pred_val_lr = clf_lr.predict(val_X)\nprint('Accuracy on Validation set :',accuracy_score(val_y, y_pred_val_lr))\nprint(\"\\n\")\nprint(classification_report(val_y, y_pred_val_lr))","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:52.785092Z","iopub.execute_input":"2022-11-05T12:42:52.785807Z","iopub.status.idle":"2022-11-05T12:42:52.836436Z","shell.execute_reply.started":"2022-11-05T12:42:52.785764Z","shell.execute_reply":"2022-11-05T12:42:52.835206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Support Vector Classifier**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import SGDClassifier","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:52.838197Z","iopub.execute_input":"2022-11-05T12:42:52.838913Z","iopub.status.idle":"2022-11-05T12:42:52.845130Z","shell.execute_reply.started":"2022-11-05T12:42:52.838867Z","shell.execute_reply":"2022-11-05T12:42:52.843558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgd = SGDClassifier(loss='hinge', penalty='l2',alpha=1e-3, random_state=42, max_iter=5, tol=None)\nsgd.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:52.847682Z","iopub.execute_input":"2022-11-05T12:42:52.848157Z","iopub.status.idle":"2022-11-05T12:42:53.007442Z","shell.execute_reply.started":"2022-11-05T12:42:52.848113Z","shell.execute_reply":"2022-11-05T12:42:53.006267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ny_pred_val_sgd = sgd.predict(val_X)\nprint('Accuracy on Validation set :',accuracy_score(val_y, y_pred_val_sgd))\nprint(\"\\n\")\nprint(classification_report(val_y, y_pred_val_sgd))","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:53.009121Z","iopub.execute_input":"2022-11-05T12:42:53.009877Z","iopub.status.idle":"2022-11-05T12:42:53.059914Z","shell.execute_reply.started":"2022-11-05T12:42:53.009830Z","shell.execute_reply":"2022-11-05T12:42:53.058640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Decision Tree Classifier**","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:53.062143Z","iopub.execute_input":"2022-11-05T12:42:53.062613Z","iopub.status.idle":"2022-11-05T12:42:53.068921Z","shell.execute_reply.started":"2022-11-05T12:42:53.062560Z","shell.execute_reply":"2022-11-05T12:42:53.067459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dc = DecisionTreeClassifier(random_state=0)\ndc.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:42:53.070975Z","iopub.execute_input":"2022-11-05T12:42:53.071426Z","iopub.status.idle":"2022-11-05T12:43:08.908717Z","shell.execute_reply.started":"2022-11-05T12:42:53.071381Z","shell.execute_reply":"2022-11-05T12:43:08.907859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ny_pred_val_dc = dc.predict(val_X)\nprint('Accuracy on Validation set :',accuracy_score(val_y, y_pred_val_dc))\nprint(\"\\n\")\nprint(classification_report(val_y, y_pred_val_dc))","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:43:08.909996Z","iopub.execute_input":"2022-11-05T12:43:08.910981Z","iopub.status.idle":"2022-11-05T12:43:08.942843Z","shell.execute_reply.started":"2022-11-05T12:43:08.910945Z","shell.execute_reply":"2022-11-05T12:43:08.941515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:44:05.743144Z","iopub.execute_input":"2022-11-05T12:44:05.743852Z","iopub.status.idle":"2022-11-05T12:44:06.576377Z","shell.execute_reply.started":"2022-11-05T12:44:05.743812Z","shell.execute_reply":"2022-11-05T12:44:06.575208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm = LGBMClassifier()\nlgbm.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:44:21.759837Z","iopub.execute_input":"2022-11-05T12:44:21.761069Z","iopub.status.idle":"2022-11-05T12:44:25.065331Z","shell.execute_reply.started":"2022-11-05T12:44:21.761029Z","shell.execute_reply":"2022-11-05T12:44:25.064320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ny_pred_val_lgbm = lgbm.predict(val_X)\nprint('Accuracy on Validation set :',accuracy_score(val_y, y_pred_val_lgbm))\nprint(\"\\n\")\nprint(classification_report(val_y, y_pred_val_lgbm))","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:44:37.634826Z","iopub.execute_input":"2022-11-05T12:44:37.635198Z","iopub.status.idle":"2022-11-05T12:44:37.707135Z","shell.execute_reply.started":"2022-11-05T12:44:37.635167Z","shell.execute_reply":"2022-11-05T12:44:37.705952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Listing Accuracies on Validation Data**","metadata":{}},{"cell_type":"code","source":"print('\\nAccuracy of Logistic Regression :',accuracy_score(val_y, y_pred_val_lr))\nprint('\\nAccuracy of Support Vector :',accuracy_score(val_y, y_pred_val_sgd))\nprint('\\nAccuracy of Decision Tree :',accuracy_score(val_y, y_pred_val_dc))\nprint('\\nAccuracy Light GBM Classifier :',accuracy_score(val_y, y_pred_val_lgbm))","metadata":{"execution":{"iopub.status.busy":"2022-11-05T12:45:13.354451Z","iopub.execute_input":"2022-11-05T12:45:13.354855Z","iopub.status.idle":"2022-11-05T12:45:13.365752Z","shell.execute_reply.started":"2022-11-05T12:45:13.354823Z","shell.execute_reply":"2022-11-05T12:45:13.364526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inferences","metadata":{}},{"cell_type":"markdown","source":"The best accuracy is obtained using Light GBM Classifier.","metadata":{}}]}