{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sn\n\nimport optuna\n\nfrom sklearn.model_selection import train_test_split\nimport sklearn.metrics\n\nfrom xgboost import XGBClassifier\n\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\n\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:07:44.793260Z","iopub.execute_input":"2022-06-25T13:07:44.794145Z","iopub.status.idle":"2022-06-25T13:07:50.840855Z","shell.execute_reply.started":"2022-06-25T13:07:44.793705Z","shell.execute_reply":"2022-06-25T13:07:50.839788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_train_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # FILL NAN\n    df = df.fillna(0) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_train_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:19.434571Z","iopub.execute_input":"2022-06-25T13:09:19.435099Z","iopub.status.idle":"2022-06-25T13:09:21.682155Z","shell.execute_reply.started":"2022-06-25T13:09:19.435053Z","shell.execute_reply":"2022-06-25T13:09:21.681150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:26.109002Z","iopub.execute_input":"2022-06-25T13:09:26.109505Z","iopub.status.idle":"2022-06-25T13:09:26.571103Z","shell.execute_reply.started":"2022-06-25T13:09:26.109462Z","shell.execute_reply":"2022-06-25T13:09:26.570062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:27.320939Z","iopub.execute_input":"2022-06-25T13:09:27.321455Z","iopub.status.idle":"2022-06-25T13:09:28.922738Z","shell.execute_reply.started":"2022-06-25T13:09:27.321414Z","shell.execute_reply":"2022-06-25T13:09:28.921044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:28.927436Z","iopub.execute_input":"2022-06-25T13:09:28.930206Z","iopub.status.idle":"2022-06-25T13:09:30.633386Z","shell.execute_reply.started":"2022-06-25T13:09:28.930160Z","shell.execute_reply":"2022-06-25T13:09:30.632436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:30.639092Z","iopub.execute_input":"2022-06-25T13:09:30.641466Z","iopub.status.idle":"2022-06-25T13:09:32.336407Z","shell.execute_reply.started":"2022-06-25T13:09:30.641423Z","shell.execute_reply":"2022-06-25T13:09:32.335460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = train.to_pandas()\ndel train\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:32.341481Z","iopub.execute_input":"2022-06-25T13:09:32.343997Z","iopub.status.idle":"2022-06-25T13:09:38.242792Z","shell.execute_reply.started":"2022-06-25T13:09:32.343934Z","shell.execute_reply":"2022-06-25T13:09:38.241586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, test_df = train_test_split(train_pd, test_size=0.25, stratify=train_pd['target'])\ndel train_pd\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:38.245368Z","iopub.execute_input":"2022-06-25T13:09:38.245747Z","iopub.status.idle":"2022-06-25T13:09:43.297859Z","shell.execute_reply.started":"2022-06-25T13:09:38.245709Z","shell.execute_reply":"2022-06-25T13:09:43.296905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df),len(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:43.299525Z","iopub.execute_input":"2022-06-25T13:09:43.300137Z","iopub.status.idle":"2022-06-25T13:09:43.307051Z","shell.execute_reply.started":"2022-06-25T13:09:43.300099Z","shell.execute_reply":"2022-06-25T13:09:43.306298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_df.drop(['customer_ID', 'target'], axis=1)\nX_test = test_df.drop(['customer_ID', 'target'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:43.310016Z","iopub.execute_input":"2022-06-25T13:09:43.310824Z","iopub.status.idle":"2022-06-25T13:09:44.150883Z","shell.execute_reply.started":"2022-06-25T13:09:43.310783Z","shell.execute_reply":"2022-06-25T13:09:44.149906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:44.152552Z","iopub.execute_input":"2022-06-25T13:09:44.153282Z","iopub.status.idle":"2022-06-25T13:09:44.292432Z","shell.execute_reply.started":"2022-06-25T13:09:44.153240Z","shell.execute_reply":"2022-06-25T13:09:44.291390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['target']\ny_test = test_df['target']","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:44.294311Z","iopub.execute_input":"2022-06-25T13:09:44.295070Z","iopub.status.idle":"2022-06-25T13:09:44.301448Z","shell.execute_reply.started":"2022-06-25T13:09:44.295021Z","shell.execute_reply":"2022-06-25T13:09:44.300079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:44.305367Z","iopub.execute_input":"2022-06-25T13:09:44.306106Z","iopub.status.idle":"2022-06-25T13:09:44.321460Z","shell.execute_reply.started":"2022-06-25T13:09:44.306062Z","shell.execute_reply":"2022-06-25T13:09:44.320352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df, test_df\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:09:44.326603Z","iopub.execute_input":"2022-06-25T13:09:44.327079Z","iopub.status.idle":"2022-06-25T13:09:44.541241Z","shell.execute_reply.started":"2022-06-25T13:09:44.327035Z","shell.execute_reply":"2022-06-25T13:09:44.540040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# optuna\n\ndef objective(trial):\n    \n    param = {\n        'booster':'gbtree',\n        'tree_method':'gpu_hist', \n        \"objective\": \"binary:logistic\",\n        'lambda': trial.suggest_loguniform(\n            'lambda', 0.01, 1.0\n        ),\n        'alpha': trial.suggest_loguniform(\n            'alpha', 5, 20.0\n        ),\n        'colsample_bytree': trial.suggest_float(\n            'colsample_bytree', 0.3,0.9,step=0.1\n        ),\n        'subsample': trial.suggest_float(\n            'subsample', 0.5,1,step=0.1\n        ),\n        'learning_rate': trial.suggest_float(\n            'learning_rate', 0.01,0.1,step=0.001\n        ),\n        'n_estimators': trial.suggest_int(\n            \"n_estimators\", 800,1200,20\n        ),\n        'max_depth': trial.suggest_int(\n            'max_depth', 4,12,1\n        ),\n        'random_state': 99,\n        'min_child_weight': trial.suggest_int(\n            'min_child_weight', 64,256,1\n        ),\n    }\n    \n    model = XGBClassifier(**param, enable_categorical = True) \n    \n    model.fit(X_train,y_train)\n    \n    preds = pd.DataFrame(model.predict(X_test))\n    \n    accuracy = sklearn.metrics.accuracy_score(pd.DataFrame(y_test.reset_index()['target']),preds)\n    \n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:21:20.740968Z","iopub.execute_input":"2022-06-25T13:21:20.741711Z","iopub.status.idle":"2022-06-25T13:21:20.753210Z","shell.execute_reply.started":"2022-06-25T13:21:20.741676Z","shell.execute_reply":"2022-06-25T13:21:20.750532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials= 200)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:21:25.660182Z","iopub.execute_input":"2022-06-25T13:21:25.660564Z","iopub.status.idle":"2022-06-25T13:22:11.216535Z","shell.execute_reply.started":"2022-06-25T13:21:25.660535Z","shell.execute_reply":"2022-06-25T13:22:11.215699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = study.best_trial.params\nbest_params['tree_method'] = 'gpu_hist'\nbest_params['booster'] = 'gbtree'\nprint(best_params)","metadata":{"execution":{"iopub.status.busy":"2022-06-25T13:22:11.218209Z","iopub.execute_input":"2022-06-25T13:22:11.218763Z","iopub.status.idle":"2022-06-25T13:22:11.360631Z","shell.execute_reply.started":"2022-06-25T13:22:11.218724Z","shell.execute_reply":"2022-06-25T13:22:11.359249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = XGBClassifier(**best_params,enable_categorical = True)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:27:02.631066Z","iopub.execute_input":"2022-06-07T19:27:02.631855Z","iopub.status.idle":"2022-06-07T19:27:02.64393Z","shell.execute_reply.started":"2022-06-07T19:27:02.631798Z","shell.execute_reply":"2022-06-07T19:27:02.642705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:27:02.64531Z","iopub.execute_input":"2022-06-07T19:27:02.646496Z","iopub.status.idle":"2022-06-07T19:28:04.808314Z","shell.execute_reply.started":"2022-06-07T19:27:02.646434Z","shell.execute_reply":"2022-06-07T19:28:04.807354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train,X_test,y_train,y_test\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:28:04.809882Z","iopub.execute_input":"2022-06-07T19:28:04.81034Z","iopub.status.idle":"2022-06-07T19:28:04.96062Z","shell.execute_reply.started":"2022-06-07T19:28:04.8103Z","shell.execute_reply":"2022-06-07T19:28:04.959396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(final_model, \"xgb_classifier_v1.h5\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def read_test_file(path = '', usecols = None):\n#     # LOAD DATAFRAME\n#     if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n#     else: df = cudf.read_parquet(path)\n#     # REDUCE DTYPE FOR CUSTOMER AND DATE\n#     #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n#     df.S_2 = cudf.to_datetime( df.S_2 )\n#     # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n#     #df = df.sort_values(['customer_ID','S_2'])\n#     #df = df.reset_index(drop=True)\n#     # FILL NAN\n#     df = df.fillna(0) \n#     print('shape of data:', df.shape)\n    \n#     return df\n\n# print('Reading test data...')\n# TEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n# test = read_test_file(path = TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:28:04.962526Z","iopub.execute_input":"2022-06-07T19:28:04.963212Z","iopub.status.idle":"2022-06-07T19:28:44.864946Z","shell.execute_reply.started":"2022-06-07T19:28:04.963165Z","shell.execute_reply":"2022-06-07T19:28:44.863764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-25T07:19:02.030779Z","iopub.execute_input":"2022-06-25T07:19:02.031849Z","iopub.status.idle":"2022-06-25T07:19:02.061479Z","shell.execute_reply.started":"2022-06-25T07:19:02.031729Z","shell.execute_reply":"2022-06-25T07:19:02.060543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test = process_and_feature_engineer(test)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:28:45.14988Z","iopub.execute_input":"2022-06-07T19:28:45.15065Z","iopub.status.idle":"2022-06-07T19:28:53.331966Z","shell.execute_reply.started":"2022-06-07T19:28:45.150604Z","shell.execute_reply":"2022-06-07T19:28:53.330937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test['prediction'] = final_model.predict_proba(test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:28:53.333592Z","iopub.execute_input":"2022-06-07T19:28:53.335564Z","iopub.status.idle":"2022-06-07T19:28:54.624865Z","shell.execute_reply.started":"2022-06-07T19:28:53.335521Z","shell.execute_reply":"2022-06-07T19:28:54.62385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final = pd.DataFrame(test['prediction'].to_pandas())","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:28:54.626475Z","iopub.execute_input":"2022-06-07T19:28:54.626978Z","iopub.status.idle":"2022-06-07T19:28:54.998313Z","shell.execute_reply.started":"2022-06-07T19:28:54.626865Z","shell.execute_reply":"2022-06-07T19:28:54.997293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final.to_csv(\"submission.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T19:28:54.999879Z","iopub.execute_input":"2022-06-07T19:28:55.000356Z","iopub.status.idle":"2022-06-07T19:28:58.360365Z","shell.execute_reply.started":"2022-06-07T19:28:55.000293Z","shell.execute_reply":"2022-06-07T19:28:58.359343Z"},"trusted":true},"execution_count":null,"outputs":[]}]}