{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom xgboost import XGBClassifier\nfrom mlxtend.classifier import StackingCVClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-16T20:49:10.390942Z","iopub.execute_input":"2022-06-16T20:49:10.392215Z","iopub.status.idle":"2022-06-16T20:49:11.604386Z","shell.execute_reply.started":"2022-06-16T20:49:10.392103Z","shell.execute_reply":"2022-06-16T20:49:11.603483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_feather('../input/amexfeather/train_data.ftr')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:13.612024Z","iopub.execute_input":"2022-06-16T20:49:13.612540Z","iopub.status.idle":"2022-06-16T20:49:40.896548Z","shell.execute_reply.started":"2022-06-16T20:49:13.612496Z","shell.execute_reply":"2022-06-16T20:49:40.894173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:40.900434Z","iopub.execute_input":"2022-06-16T20:49:40.901815Z","iopub.status.idle":"2022-06-16T20:49:40.910696Z","shell.execute_reply.started":"2022-06-16T20:49:40.901758Z","shell.execute_reply":"2022-06-16T20:49:40.909433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train =  (train\n            .groupby('customer_ID')\n            .tail(1)\n            .set_index('customer_ID', drop=True)\n            .sort_index()\n            .drop(['S_2'], axis='columns'))","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:40.912600Z","iopub.execute_input":"2022-06-16T20:49:40.913126Z","iopub.status.idle":"2022-06-16T20:49:45.172792Z","shell.execute_reply.started":"2022-06-16T20:49:40.913085Z","shell.execute_reply":"2022-06-16T20:49:45.171602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:45.175176Z","iopub.execute_input":"2022-06-16T20:49:45.175570Z","iopub.status.idle":"2022-06-16T20:49:45.184187Z","shell.execute_reply.started":"2022-06-16T20:49:45.175537Z","shell.execute_reply":"2022-06-16T20:49:45.183058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = train.columns.to_list()\ncategory_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnumerical_cols = [col for col in cols if col not in category_cols + ['target']]","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:45.186143Z","iopub.execute_input":"2022-06-16T20:49:45.186715Z","iopub.status.idle":"2022-06-16T20:49:45.194387Z","shell.execute_reply.started":"2022-06-16T20:49:45.186665Z","shell.execute_reply":"2022-06-16T20:49:45.193609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train[category_cols + numerical_cols]\ny = train['target']\n\nX.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:45.195770Z","iopub.execute_input":"2022-06-16T20:49:45.196250Z","iopub.status.idle":"2022-06-16T20:49:45.733617Z","shell.execute_reply.started":"2022-06-16T20:49:45.196205Z","shell.execute_reply":"2022-06-16T20:49:45.732599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nenc = OrdinalEncoder()\nX[category_cols] = enc.fit_transform(X[category_cols])","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:45.735221Z","iopub.execute_input":"2022-06-16T20:49:45.735585Z","iopub.status.idle":"2022-06-16T20:49:46.719567Z","shell.execute_reply.started":"2022-06-16T20:49:45.735555Z","shell.execute_reply":"2022-06-16T20:49:46.718367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:46.721131Z","iopub.execute_input":"2022-06-16T20:49:46.721537Z","iopub.status.idle":"2022-06-16T20:49:46.738474Z","shell.execute_reply.started":"2022-06-16T20:49:46.721500Z","shell.execute_reply":"2022-06-16T20:49:46.737502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.1, stratify=y)\nX_train.shape, X_test.shape, y_train.shape, y_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:46.740002Z","iopub.execute_input":"2022-06-16T20:49:46.741396Z","iopub.status.idle":"2022-06-16T20:49:48.992273Z","shell.execute_reply.started":"2022-06-16T20:49:46.741336Z","shell.execute_reply":"2022-06-16T20:49:48.991145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm = LGBMClassifier(\n    n_estimators=10000,\n    random_state=242,\n    extra_trees=True\n)\n\nxgboost = XGBClassifier(learning_rate=0.01,n_estimators=3500,\n                                     min_child_weight=0,\n                                     gamma=0, subsample=0.7,\n                                     colsample_bytree=0.7,\n                                     objective='binary:logistic', nthread=-1,\n                                     scale_pos_weight=1, seed=27,\n                                     reg_alpha=0.00006)\n\n'''\nclf = StackingCVClassifier(classifiers=(lgbm, xgboost),\n                                meta_classifier=lgbm,\n                                use_features_in_secondary=True)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:48.994527Z","iopub.execute_input":"2022-06-16T20:49:48.994916Z","iopub.status.idle":"2022-06-16T20:49:49.003902Z","shell.execute_reply.started":"2022-06-16T20:49:48.994883Z","shell.execute_reply":"2022-06-16T20:49:49.002368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm.fit(\n    X_train, y_train, \n    eval_set=[(X_test,y_test)],\n    callbacks=[early_stopping(50), log_evaluation(0)]\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:49:49.006003Z","iopub.execute_input":"2022-06-16T20:49:49.006655Z","iopub.status.idle":"2022-06-16T20:51:11.499583Z","shell.execute_reply.started":"2022-06-16T20:49:49.006594Z","shell.execute_reply":"2022-06-16T20:51:11.498841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgboost.fit(\n    X_train, y_train,\n    eval_set=[(X_test, y_test)],\n    early_stopping_rounds=50,\n    verbose=False\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T20:59:32.437650Z","iopub.execute_input":"2022-06-16T20:59:32.438092Z","iopub.status.idle":"2022-06-16T23:51:57.021453Z","shell.execute_reply.started":"2022-06-16T20:59:32.438054Z","shell.execute_reply":"2022-06-16T23:51:57.019811Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pd.DataFrame(y_test.copy(deep=True))\ny_pred = y_pred.rename(columns={'target':'prediction'})\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-16T23:52:35.891716Z","iopub.execute_input":"2022-06-16T23:52:35.892155Z","iopub.status.idle":"2022-06-16T23:52:35.915633Z","shell.execute_reply.started":"2022-06-16T23:52:35.892104Z","shell.execute_reply":"2022-06-16T23:52:35.914514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Blending\n\ny_pred['prediction'] = 0.4 * lgbm.predict_proba(X_test)[:,1] + 0.6 * xgboost.predict_proba(X_test)[:,1]\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:00:51.058788Z","iopub.execute_input":"2022-06-17T00:00:51.059323Z","iopub.status.idle":"2022-06-17T00:00:54.300703Z","shell.execute_reply.started":"2022-06-17T00:00:51.059285Z","shell.execute_reply":"2022-06-17T00:00:54.299637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:00:56.962889Z","iopub.execute_input":"2022-06-17T00:00:56.964052Z","iopub.status.idle":"2022-06-17T00:00:56.969168Z","shell.execute_reply.started":"2022-06-17T00:00:56.964008Z","shell.execute_reply":"2022-06-17T00:00:56.967869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:01:00.109680Z","iopub.execute_input":"2022-06-17T00:01:00.110335Z","iopub.status.idle":"2022-06-17T00:01:00.319529Z","shell.execute_reply.started":"2022-06-17T00:01:00.110284Z","shell.execute_reply":"2022-06-17T00:01:00.318128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, X, y, X_test, X_train, y_train, y_test, y_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:01:16.794584Z","iopub.execute_input":"2022-06-17T00:01:16.795098Z","iopub.status.idle":"2022-06-17T00:01:16.800484Z","shell.execute_reply.started":"2022-06-17T00:01:16.795062Z","shell.execute_reply":"2022-06-17T00:01:16.799616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_feather('../input/amexfeather/test_data.ftr')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:01:20.900672Z","iopub.execute_input":"2022-06-17T00:01:20.901320Z","iopub.status.idle":"2022-06-17T00:02:09.100125Z","shell.execute_reply.started":"2022-06-17T00:01:20.901283Z","shell.execute_reply":"2022-06-17T00:02:09.098046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test =  (\n    test\n    .groupby('customer_ID')\n    .tail(1)\n    .set_index('customer_ID', drop=True)\n    .sort_index()\n    .drop(['S_2'], axis='columns')\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:02:15.065914Z","iopub.execute_input":"2022-06-17T00:02:15.066380Z","iopub.status.idle":"2022-06-17T00:02:24.546375Z","shell.execute_reply.started":"2022-06-17T00:02:15.066345Z","shell.execute_reply":"2022-06-17T00:02:24.545255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[category_cols] = enc.transform(test[category_cols])","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:02:25.911687Z","iopub.execute_input":"2022-06-17T00:02:25.912204Z","iopub.status.idle":"2022-06-17T00:02:27.511217Z","shell.execute_reply.started":"2022-06-17T00:02:25.912164Z","shell.execute_reply":"2022-06-17T00:02:27.509650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Blending\n\ntest['prediction'] = 0.4 * lgbm.predict_proba(test[category_cols + numerical_cols])[:,1] + 0.6 * xgboost.predict_proba(test[category_cols + numerical_cols])[:,1]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:03:33.199947Z","iopub.execute_input":"2022-06-17T00:03:33.201687Z","iopub.status.idle":"2022-06-17T00:04:42.993658Z","shell.execute_reply.started":"2022-06-17T00:03:33.201631Z","shell.execute_reply":"2022-06-17T00:04:42.992746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"].to_csv(\"submission.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T00:04:59.101852Z","iopub.execute_input":"2022-06-17T00:04:59.102748Z","iopub.status.idle":"2022-06-17T00:05:03.870147Z","shell.execute_reply.started":"2022-06-17T00:04:59.102711Z","shell.execute_reply":"2022-06-17T00:05:03.869134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}