{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport os\nimport warnings\n\nimport numpy as np\nimport pandas as pd\n\nimport gc  # Garbage collector\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-24T07:25:58.164504Z","iopub.execute_input":"2022-06-24T07:25:58.16533Z","iopub.status.idle":"2022-06-24T07:25:58.170468Z","shell.execute_reply.started":"2022-06-24T07:25:58.165287Z","shell.execute_reply":"2022-06-24T07:25:58.169431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### For EDA, refer: https://www.kaggle.com/code/awaldeep/first-look-eda/data","metadata":{}},{"cell_type":"markdown","source":"### Data pre-processing","metadata":{}},{"cell_type":"code","source":"# Reading feather format data(memory efficient, available on kaggle: https://www.kaggle.com/datasets/munumbutt/amexfeather) \ntrain_raw = pd.read_feather('../input/amexfeather/train_data.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:25:58.735414Z","iopub.execute_input":"2022-06-24T07:25:58.736046Z","iopub.status.idle":"2022-06-24T07:26:15.226Z","shell.execute_reply.started":"2022-06-24T07:25:58.736006Z","shell.execute_reply":"2022-06-24T07:26:15.222965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.226867Z","iopub.status.idle":"2022-06-24T07:26:15.227903Z","shell.execute_reply.started":"2022-06-24T07:26:15.227591Z","shell.execute_reply":"2022-06-24T07:26:15.227623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.229182Z","iopub.status.idle":"2022-06-24T07:26:15.230178Z","shell.execute_reply.started":"2022-06-24T07:26:15.229883Z","shell.execute_reply":"2022-06-24T07:26:15.229913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing values\ntmp = train_raw.isna().sum().mul(100).div(len(train_raw)).sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.231311Z","iopub.status.idle":"2022-06-24T07:26:15.231868Z","shell.execute_reply.started":"2022-06-24T07:26:15.231608Z","shell.execute_reply":"2022-06-24T07:26:15.231634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Handling missing values","metadata":{}},{"cell_type":"code","source":"# dropping columns with missing values >70%\nmissingDF = pd.DataFrame(tmp).reset_index()\ndrop_cols = missingDF[missingDF[0]>70][\"index\"].values\nprint(drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.233577Z","iopub.status.idle":"2022-06-24T07:26:15.234491Z","shell.execute_reply.started":"2022-06-24T07:26:15.234132Z","shell.execute_reply":"2022-06-24T07:26:15.234162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.235738Z","iopub.status.idle":"2022-06-24T07:26:15.236295Z","shell.execute_reply.started":"2022-06-24T07:26:15.236029Z","shell.execute_reply":"2022-06-24T07:26:15.236055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw.drop(columns = drop_cols,axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.238008Z","iopub.status.idle":"2022-06-24T07:26:15.238556Z","shell.execute_reply.started":"2022-06-24T07:26:15.238272Z","shell.execute_reply":"2022-06-24T07:26:15.238297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For categorical columns\ncols = train_raw.columns\nnum_cols = train_raw._get_numeric_data().columns\n\ncategorical_columns = list(set(cols) - set(num_cols))\nfiltered_categorical_columns = list(set(train_raw[categorical_columns])-{\"S_2\",\"customer_ID\"})","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.239606Z","iopub.status.idle":"2022-06-24T07:26:15.240833Z","shell.execute_reply.started":"2022-06-24T07:26:15.240563Z","shell.execute_reply":"2022-06-24T07:26:15.24059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw[filtered_categorical_columns].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.242128Z","iopub.status.idle":"2022-06-24T07:26:15.242714Z","shell.execute_reply.started":"2022-06-24T07:26:15.242455Z","shell.execute_reply":"2022-06-24T07:26:15.242482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw[filtered_categorical_columns].isna().sum().mul(100).div(len(train_raw))","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.243846Z","iopub.status.idle":"2022-06-24T07:26:15.244224Z","shell.execute_reply.started":"2022-06-24T07:26:15.244046Z","shell.execute_reply":"2022-06-24T07:26:15.244063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in filtered_categorical_columns:\n    print(train_raw[i].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.2475Z","iopub.status.idle":"2022-06-24T07:26:15.247928Z","shell.execute_reply.started":"2022-06-24T07:26:15.24775Z","shell.execute_reply":"2022-06-24T07:26:15.247769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nimputer=SimpleImputer(strategy=\"most_frequent\")\ntransformed_df = pd.DataFrame(imputer.fit_transform(train_raw[filtered_categorical_columns]),columns = filtered_categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.249044Z","iopub.status.idle":"2022-06-24T07:26:15.249437Z","shell.execute_reply.started":"2022-06-24T07:26:15.249239Z","shell.execute_reply":"2022-06-24T07:26:15.249256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw[filtered_categorical_columns] = transformed_df[filtered_categorical_columns]","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.25104Z","iopub.status.idle":"2022-06-24T07:26:15.251443Z","shell.execute_reply.started":"2022-06-24T07:26:15.251241Z","shell.execute_reply":"2022-06-24T07:26:15.251258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For numeric columns\nnumeric_columns = train_raw.select_dtypes(np.number).columns\ntrain_raw[numeric_columns] = train_raw[numeric_columns].fillna(train_raw[numeric_columns].mean())","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.252804Z","iopub.status.idle":"2022-06-24T07:26:15.253453Z","shell.execute_reply.started":"2022-06-24T07:26:15.253211Z","shell.execute_reply":"2022-06-24T07:26:15.253232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_raw.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.255067Z","iopub.status.idle":"2022-06-24T07:26:15.255497Z","shell.execute_reply.started":"2022-06-24T07:26:15.255281Z","shell.execute_reply":"2022-06-24T07:26:15.255299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling date column\n\ntrain_raw[\"S_2_day\"] = train_raw[\"S_2\"].dt.day\ntrain_raw[\"S_2_month\"] = train_raw[\"S_2\"].dt.month\ntrain_raw[\"S_2_year\"] = train_raw[\"S_2\"].dt.year\n","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.256765Z","iopub.status.idle":"2022-06-24T07:26:15.257137Z","shell.execute_reply.started":"2022-06-24T07:26:15.256964Z","shell.execute_reply":"2022-06-24T07:26:15.256982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# considering only one data point per customer (latest one) as time series is not being used\ntrain_raw = train_raw.groupby(['customer_ID']).nth(-1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.258984Z","iopub.status.idle":"2022-06-24T07:26:15.259411Z","shell.execute_reply.started":"2022-06-24T07:26:15.259203Z","shell.execute_reply":"2022-06-24T07:26:15.259221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop S_2\ntrain_raw.drop(columns=[\"S_2\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.260917Z","iopub.status.idle":"2022-06-24T07:26:15.261394Z","shell.execute_reply.started":"2022-06-24T07:26:15.261184Z","shell.execute_reply":"2022-06-24T07:26:15.261203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting pandas \"categorical\" dtype to numeric\ncols = [\"D_68\", \"B_30\", \"B_38\", \"D_114\", \"D_116\", \"D_117\", \"D_120\", \"D_126\"]\ntrain_raw[cols] = train_raw[cols].apply(pd.to_numeric, errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.262676Z","iopub.status.idle":"2022-06-24T07:26:15.26354Z","shell.execute_reply.started":"2022-06-24T07:26:15.263245Z","shell.execute_reply":"2022-06-24T07:26:15.263278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom xgboost import XGBClassifier\nimport xgboost as xgb\nfrom datetime import datetime, timedelta","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.264678Z","iopub.status.idle":"2022-06-24T07:26:15.26504Z","shell.execute_reply.started":"2022-06-24T07:26:15.26487Z","shell.execute_reply":"2022-06-24T07:26:15.264887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/inversion/amex-competition-metric-python\n\ndef amex_metric_official(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.26652Z","iopub.status.idle":"2022-06-24T07:26:15.267749Z","shell.execute_reply.started":"2022-06-24T07:26:15.267431Z","shell.execute_reply":"2022-06-24T07:26:15.267456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_raw.drop(columns=[\"target\"],axis=1)\ny = train_raw[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.269077Z","iopub.status.idle":"2022-06-24T07:26:15.269482Z","shell.execute_reply.started":"2022-06-24T07:26:15.269277Z","shell.execute_reply":"2022-06-24T07:26:15.269295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.33,random_state=100)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.270807Z","iopub.status.idle":"2022-06-24T07:26:15.271184Z","shell.execute_reply.started":"2022-06-24T07:26:15.271012Z","shell.execute_reply":"2022-06-24T07:26:15.27103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label encoding\nfrom sklearn.preprocessing import OrdinalEncoder\n\ncategorical_columns = [\"D_63\",\"D_64\"]\n\noe = OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-999)\noe.fit(X_train[categorical_columns])\n\nX_train_enc = oe.transform(X_train[categorical_columns])\nX_test_enc = oe.transform(X_test[categorical_columns])\n\nX_train[categorical_columns] = X_train_enc\nX_test[categorical_columns] = X_test_enc","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.272401Z","iopub.status.idle":"2022-06-24T07:26:15.27279Z","shell.execute_reply.started":"2022-06-24T07:26:15.272609Z","shell.execute_reply":"2022-06-24T07:26:15.272627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train.to_csv(\"x_train.csv\", index=False)\n# X_test.to_csv(\"x_test.csv\", index=False)\n# y_train.to_csv(\"y_train.csv\", index=False)\n# y_test.to_csv(\"y_test.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.273956Z","iopub.status.idle":"2022-06-24T07:26:15.274329Z","shell.execute_reply.started":"2022-06-24T07:26:15.274157Z","shell.execute_reply":"2022-06-24T07:26:15.274175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_classifier = XGBClassifier(objective='binary:logistic', \n                      n_estimators=200,\n                      eta=0.2,\n                      seed=12,\n                      learning_rate=0.02,\n                      use_label_encoder=False,\n                      eval_metric='aucpr',                      \n#                       early_stopping_rounds=10,tree_method='gpu_hist',enable_categorical=True\n                            )\nxgb_classifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.275446Z","iopub.status.idle":"2022-06-24T07:26:15.276088Z","shell.execute_reply.started":"2022-06-24T07:26:15.275858Z","shell.execute_reply":"2022-06-24T07:26:15.275879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = xgb_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:26:15.277258Z","iopub.status.idle":"2022-06-24T07:26:15.277647Z","shell.execute_reply.started":"2022-06-24T07:26:15.277472Z","shell.execute_reply":"2022-06-24T07:26:15.27749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_prob = xgb_classifier.predict_proba(X_test)[:,1]\n","metadata":{"execution":{"iopub.status.busy":"2022-06-24T06:59:21.891528Z","iopub.execute_input":"2022-06-24T06:59:21.895714Z","iopub.status.idle":"2022-06-24T06:59:22.587096Z","shell.execute_reply.started":"2022-06-24T06:59:21.895667Z","shell.execute_reply":"2022-06-24T06:59:22.586335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns=[\"target\"])\ny_pred = pd.DataFrame(y_pred, columns=[\"prediction\"])\ny_pred_prob = pd.DataFrame(y_pred_prob, columns=[\"prediction\"])","metadata":{"execution":{"iopub.status.busy":"2022-06-24T06:59:22.591725Z","iopub.execute_input":"2022-06-24T06:59:22.594098Z","iopub.status.idle":"2022-06-24T06:59:22.607434Z","shell.execute_reply.started":"2022-06-24T06:59:22.594055Z","shell.execute_reply":"2022-06-24T06:59:22.606369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # computing metric score\namex_metric_official(y_test, y_pred_prob)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T06:59:22.609756Z","iopub.execute_input":"2022-06-24T06:59:22.61014Z","iopub.status.idle":"2022-06-24T06:59:23.318837Z","shell.execute_reply.started":"2022-06-24T06:59:22.610109Z","shell.execute_reply":"2022-06-24T06:59:23.317864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute accuracy\naccuracy = metrics.accuracy_score(y_test[\"target\"], y_pred[\"prediction\"])\nprint(f'accuracy: {accuracy: .2%}')","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:19:54.582407Z","iopub.execute_input":"2022-06-24T07:19:54.582864Z","iopub.status.idle":"2022-06-24T07:19:54.605251Z","shell.execute_reply.started":"2022-06-24T07:19:54.582831Z","shell.execute_reply":"2022-06-24T07:19:54.604349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(xgb_classifier, \"xgb_classifier_v1.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-06-23T16:14:04.222624Z","iopub.execute_input":"2022-06-23T16:14:04.222977Z","iopub.status.idle":"2022-06-23T16:14:04.226604Z","shell.execute_reply.started":"2022-06-23T16:14:04.222945Z","shell.execute_reply":"2022-06-23T16:14:04.225747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(oe, \"oe.h5\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # load the model\n# import joblib\n# xgb_classifier = joblib.load(\"../input/01-starter-xgboost-implementation/xgb_classifier_v1.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-06-23T16:14:04.227701Z","iopub.execute_input":"2022-06-23T16:14:04.228036Z","iopub.status.idle":"2022-06-23T16:14:04.237553Z","shell.execute_reply.started":"2022-06-23T16:14:04.228006Z","shell.execute_reply":"2022-06-23T16:14:04.236545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission in 02. xgboost implementation","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n## DO UPVOTE !","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}