{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-26T01:21:29.518928Z","iopub.execute_input":"2022-06-26T01:21:29.519978Z","iopub.status.idle":"2022-06-26T01:21:29.530217Z","shell.execute_reply.started":"2022-06-26T01:21:29.51993Z","shell.execute_reply":"2022-06-26T01:21:29.529072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:09:53.071752Z","iopub.execute_input":"2022-06-26T01:09:53.072743Z","iopub.status.idle":"2022-06-26T01:09:54.024626Z","shell.execute_reply.started":"2022-06-26T01:09:53.072706Z","shell.execute_reply":"2022-06-26T01:09:54.023682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:11:44.369158Z","iopub.execute_input":"2022-06-26T01:11:44.369724Z","iopub.status.idle":"2022-06-26T01:11:44.385298Z","shell.execute_reply.started":"2022-06-26T01:11:44.369681Z","shell.execute_reply":"2022-06-26T01:11:44.383958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.pandas.read_feather('/kaggle/input/amexfeather/train_data.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:21:46.345638Z","iopub.execute_input":"2022-06-26T01:21:46.345993Z","iopub.status.idle":"2022-06-26T01:21:52.50093Z","shell.execute_reply.started":"2022-06-26T01:21:46.345963Z","shell.execute_reply":"2022-06-26T01:21:52.499955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:10:21.813326Z","iopub.execute_input":"2022-06-26T01:10:21.813752Z","iopub.status.idle":"2022-06-26T01:10:23.334097Z","shell.execute_reply.started":"2022-06-26T01:10:21.813717Z","shell.execute_reply":"2022-06-26T01:10:23.332975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = train_df.isna().sum().mul(100).div(len(train_df)).sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:27:05.369608Z","iopub.execute_input":"2022-06-26T01:27:05.370539Z","iopub.status.idle":"2022-06-26T01:27:10.331308Z","shell.execute_reply.started":"2022-06-26T01:27:05.37049Z","shell.execute_reply":"2022-06-26T01:27:10.330334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:28:20.827536Z","iopub.execute_input":"2022-06-26T01:28:20.828009Z","iopub.status.idle":"2022-06-26T01:28:20.846474Z","shell.execute_reply.started":"2022-06-26T01:28:20.827968Z","shell.execute_reply":"2022-06-26T01:28:20.845458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missingDF = pd.DataFrame(tmp).reset_index()\ndrop_cols = missingDF[missingDF[0]>70][\"index\"].values\nprint(drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:28:27.759002Z","iopub.execute_input":"2022-06-26T01:28:27.759405Z","iopub.status.idle":"2022-06-26T01:28:27.773912Z","shell.execute_reply.started":"2022-06-26T01:28:27.759376Z","shell.execute_reply":"2022-06-26T01:28:27.77279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(columns = drop_cols,axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:30:26.661427Z","iopub.execute_input":"2022-06-26T01:30:26.661773Z","iopub.status.idle":"2022-06-26T01:30:29.814231Z","shell.execute_reply.started":"2022-06-26T01:30:26.661743Z","shell.execute_reply":"2022-06-26T01:30:29.813077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:33:52.922611Z","iopub.execute_input":"2022-06-26T01:33:52.924149Z","iopub.status.idle":"2022-06-26T01:33:54.528461Z","shell.execute_reply.started":"2022-06-26T01:33:52.924092Z","shell.execute_reply":"2022-06-26T01:33:54.527097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For categorical columns\ncols = train_df.columns\nnum_cols = train_df._get_numeric_data().columns\n\ncategorical_columns = list(set(cols) - set(num_cols))\nfiltered_categorical_columns = list(set(train_df[categorical_columns])-{\"S_2\",\"customer_ID\"})","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:35:09.483643Z","iopub.execute_input":"2022-06-26T01:35:09.484154Z","iopub.status.idle":"2022-06-26T01:35:09.607443Z","shell.execute_reply.started":"2022-06-26T01:35:09.484117Z","shell.execute_reply":"2022-06-26T01:35:09.605932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[filtered_categorical_columns].isna().sum().mul(100).div(len(train_df))","metadata":{"execution":{"iopub.status.busy":"2022-06-26T01:37:39.919863Z","iopub.execute_input":"2022-06-26T01:37:39.920961Z","iopub.status.idle":"2022-06-26T01:37:40.030274Z","shell.execute_reply.started":"2022-06-26T01:37:39.920916Z","shell.execute_reply":"2022-06-26T01:37:40.028994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in filtered_categorical_columns:\n    print(train_df[i].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:02:27.976959Z","iopub.execute_input":"2022-06-26T02:02:27.977539Z","iopub.status.idle":"2022-06-26T02:02:28.322093Z","shell.execute_reply.started":"2022-06-26T02:02:27.977505Z","shell.execute_reply":"2022-06-26T02:02:28.321074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nimputer=SimpleImputer(strategy=\"most_frequent\")\ntransformed_df = pd.DataFrame(imputer.fit_transform(train_df[filtered_categorical_columns]),columns = filtered_categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:04:23.266866Z","iopub.execute_input":"2022-06-26T02:04:23.267307Z","iopub.status.idle":"2022-06-26T02:04:53.756629Z","shell.execute_reply.started":"2022-06-26T02:04:23.267274Z","shell.execute_reply":"2022-06-26T02:04:53.755642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[filtered_categorical_columns] = transformed_df[filtered_categorical_columns]","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:05:35.1783Z","iopub.execute_input":"2022-06-26T02:05:35.178638Z","iopub.status.idle":"2022-06-26T02:05:36.660489Z","shell.execute_reply.started":"2022-06-26T02:05:35.178611Z","shell.execute_reply":"2022-06-26T02:05:36.659499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For numeric columns\nnumeric_columns = train_df.select_dtypes(np.number).columns\ntrain_df[numeric_columns] = train_df[numeric_columns].fillna(train_df[numeric_columns].mean())","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:06:17.901393Z","iopub.execute_input":"2022-06-26T02:06:17.901815Z","iopub.status.idle":"2022-06-26T02:06:44.351062Z","shell.execute_reply.started":"2022-06-26T02:06:17.901776Z","shell.execute_reply":"2022-06-26T02:06:44.35007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:06:51.88456Z","iopub.execute_input":"2022-06-26T02:06:51.885235Z","iopub.status.idle":"2022-06-26T02:06:51.927411Z","shell.execute_reply.started":"2022-06-26T02:06:51.885191Z","shell.execute_reply":"2022-06-26T02:06:51.926374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling date column\n\ntrain_df[\"S_2_day\"] = train_df[\"S_2\"].dt.day\ntrain_df[\"S_2_month\"] = train_df[\"S_2\"].dt.month\ntrain_df[\"S_2_year\"] = train_df[\"S_2\"].dt.year","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:07:21.242072Z","iopub.execute_input":"2022-06-26T02:07:21.242592Z","iopub.status.idle":"2022-06-26T02:07:22.710012Z","shell.execute_reply.started":"2022-06-26T02:07:21.242547Z","shell.execute_reply":"2022-06-26T02:07:22.709079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# considering only one data point per customer (latest one) as time series is not being used\ntrain_df = train_df.groupby(['customer_ID']).nth(-1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:07:39.808873Z","iopub.execute_input":"2022-06-26T02:07:39.80944Z","iopub.status.idle":"2022-06-26T02:07:47.377235Z","shell.execute_reply.started":"2022-06-26T02:07:39.809405Z","shell.execute_reply":"2022-06-26T02:07:47.376261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop S_2\ntrain_df.drop(columns=[\"S_2\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:07:54.357919Z","iopub.execute_input":"2022-06-26T02:07:54.358843Z","iopub.status.idle":"2022-06-26T02:07:54.753984Z","shell.execute_reply.started":"2022-06-26T02:07:54.358796Z","shell.execute_reply":"2022-06-26T02:07:54.753054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting pandas \"categorical\" dtype to numeric\ncols = [\"D_68\", \"B_30\", \"B_38\", \"D_114\", \"D_116\", \"D_117\", \"D_120\", \"D_126\"]\ntrain_df[cols] = train_df[cols].apply(pd.to_numeric, errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:08:13.237932Z","iopub.execute_input":"2022-06-26T02:08:13.238581Z","iopub.status.idle":"2022-06-26T02:08:14.579078Z","shell.execute_reply.started":"2022-06-26T02:08:13.238545Z","shell.execute_reply":"2022-06-26T02:08:14.578069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom xgboost import XGBClassifier\nimport xgboost as xgb\nfrom datetime import datetime, timedelta","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:08:30.09076Z","iopub.execute_input":"2022-06-26T02:08:30.091135Z","iopub.status.idle":"2022-06-26T02:08:30.215088Z","shell.execute_reply.started":"2022-06-26T02:08:30.091105Z","shell.execute_reply":"2022-06-26T02:08:30.214143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/inversion/amex-competition-metric-python\n\ndef amex_metric_official(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:08:49.777129Z","iopub.execute_input":"2022-06-26T02:08:49.777485Z","iopub.status.idle":"2022-06-26T02:08:49.793456Z","shell.execute_reply.started":"2022-06-26T02:08:49.777455Z","shell.execute_reply":"2022-06-26T02:08:49.792446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(columns=[\"target\"],axis=1)\ny = train_df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:09:46.897539Z","iopub.execute_input":"2022-06-26T02:09:46.898144Z","iopub.status.idle":"2022-06-26T02:09:47.322438Z","shell.execute_reply.started":"2022-06-26T02:09:46.898102Z","shell.execute_reply":"2022-06-26T02:09:47.320351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.33,random_state=100)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:10:01.180449Z","iopub.execute_input":"2022-06-26T02:10:01.18097Z","iopub.status.idle":"2022-06-26T02:10:02.68545Z","shell.execute_reply.started":"2022-06-26T02:10:01.180938Z","shell.execute_reply":"2022-06-26T02:10:02.684363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label encoding\nfrom sklearn.preprocessing import OrdinalEncoder\n\ncategorical_columns = [\"D_63\",\"D_64\"]\n\noe = OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-999)\noe.fit(X_train[categorical_columns])\n\nX_train_enc = oe.transform(X_train[categorical_columns])\nX_test_enc = oe.transform(X_test[categorical_columns])\n\nX_train[categorical_columns] = X_train_enc\nX_test[categorical_columns] = X_test_enc","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:10:15.038676Z","iopub.execute_input":"2022-06-26T02:10:15.0391Z","iopub.status.idle":"2022-06-26T02:10:15.310227Z","shell.execute_reply.started":"2022-06-26T02:10:15.039065Z","shell.execute_reply":"2022-06-26T02:10:15.309036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_classifier = XGBClassifier(objective='binary:logistic', \n                      n_estimators=200,\n                      eta=0.2,\n                      seed=12,\n                      learning_rate=0.02,\n                      use_label_encoder=False,\n                      eval_metric='aucpr',                      \n#                       early_stopping_rounds=10,tree_method='gpu_hist',enable_categorical=True\n                            )\nxgb_classifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-26T02:10:28.586941Z","iopub.execute_input":"2022-06-26T02:10:28.587816Z","iopub.status.idle":"2022-06-26T02:27:26.35127Z","shell.execute_reply.started":"2022-06-26T02:10:28.587766Z","shell.execute_reply":"2022-06-26T02:27:26.35048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}