{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sn\n\nimport optuna\n\nfrom sklearn.model_selection import train_test_split\nimport sklearn.metrics\n\nfrom xgboost import XGBClassifier\n\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\n\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:23:50.377259Z","iopub.execute_input":"2022-06-08T19:23:50.377724Z","iopub.status.idle":"2022-06-08T19:23:52.31658Z","shell.execute_reply.started":"2022-06-08T19:23:50.377636Z","shell.execute_reply":"2022-06-08T19:23:52.315673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric_np(preds: np.ndarray, target: np.ndarray) -> float:\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n\n    g = gini / gini_max\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:23:52.321941Z","iopub.execute_input":"2022-06-08T19:23:52.324702Z","iopub.status.idle":"2022-06-08T19:23:52.3382Z","shell.execute_reply.started":"2022-06-08T19:23:52.324658Z","shell.execute_reply":"2022-06-08T19:23:52.337346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_train_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(0) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_train_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:23:52.343367Z","iopub.execute_input":"2022-06-08T19:23:52.346253Z","iopub.status.idle":"2022-06-08T19:24:11.647182Z","shell.execute_reply.started":"2022-06-08T19:23:52.346204Z","shell.execute_reply":"2022-06-08T19:24:11.646021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n    \n    \n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min','max','last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:11.653681Z","iopub.execute_input":"2022-06-08T19:24:11.654214Z","iopub.status.idle":"2022-06-08T19:24:13.227532Z","shell.execute_reply.started":"2022-06-08T19:24:11.654169Z","shell.execute_reply":"2022-06-08T19:24:13.22645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:13.229492Z","iopub.execute_input":"2022-06-08T19:24:13.230046Z","iopub.status.idle":"2022-06-08T19:24:14.858191Z","shell.execute_reply.started":"2022-06-08T19:24:13.229991Z","shell.execute_reply":"2022-06-08T19:24:14.856672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pandas = train.to_pandas()\ndel train\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:14.860222Z","iopub.execute_input":"2022-06-08T19:24:14.860755Z","iopub.status.idle":"2022-06-08T19:24:19.626864Z","shell.execute_reply.started":"2022-06-08T19:24:14.860714Z","shell.execute_reply":"2022-06-08T19:24:19.625957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, test_df = train_test_split(train_pandas, test_size=0.2, stratify=train_pandas['target'])\ndel train_pandas\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:19.628428Z","iopub.execute_input":"2022-06-08T19:24:19.628852Z","iopub.status.idle":"2022-06-08T19:24:22.311058Z","shell.execute_reply.started":"2022-06-08T19:24:19.628805Z","shell.execute_reply":"2022-06-08T19:24:22.310126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df),len(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:22.312489Z","iopub.execute_input":"2022-06-08T19:24:22.312902Z","iopub.status.idle":"2022-06-08T19:24:22.321452Z","shell.execute_reply.started":"2022-06-08T19:24:22.312861Z","shell.execute_reply":"2022-06-08T19:24:22.320566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_df.drop(['customer_ID', 'target'], axis=1)\nX_test = test_df.drop(['customer_ID', 'target'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:22.322956Z","iopub.execute_input":"2022-06-08T19:24:22.323542Z","iopub.status.idle":"2022-06-08T19:24:23.014127Z","shell.execute_reply.started":"2022-06-08T19:24:22.323497Z","shell.execute_reply":"2022-06-08T19:24:23.013138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:23.016884Z","iopub.execute_input":"2022-06-08T19:24:23.017391Z","iopub.status.idle":"2022-06-08T19:24:23.124651Z","shell.execute_reply.started":"2022-06-08T19:24:23.017346Z","shell.execute_reply":"2022-06-08T19:24:23.123677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df['target']\ny_test = test_df['target']","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:23.126146Z","iopub.execute_input":"2022-06-08T19:24:23.126793Z","iopub.status.idle":"2022-06-08T19:24:23.13284Z","shell.execute_reply.started":"2022-06-08T19:24:23.126748Z","shell.execute_reply":"2022-06-08T19:24:23.131814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:23.134566Z","iopub.execute_input":"2022-06-08T19:24:23.135199Z","iopub.status.idle":"2022-06-08T19:24:23.147027Z","shell.execute_reply.started":"2022-06-08T19:24:23.13515Z","shell.execute_reply":"2022-06-08T19:24:23.145656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df, test_df\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:24:23.148821Z","iopub.execute_input":"2022-06-08T19:24:23.149343Z","iopub.status.idle":"2022-06-08T19:24:23.301991Z","shell.execute_reply.started":"2022-06-08T19:24:23.149311Z","shell.execute_reply":"2022-06-08T19:24:23.300667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# optuna\n\ndef objective(trial):\n    \n    param = {\n        'booster':'gbtree',\n        'tree_method':'gpu_hist', \n        \"objective\": \"binary:logistic\",\n        'lambda': trial.suggest_loguniform(\n            'lambda', 1e-3, 10.0\n        ),\n        'alpha': trial.suggest_loguniform(\n            'alpha', 1e-3, 10.0\n        ),\n        'colsample_bytree': trial.suggest_float(\n            'colsample_bytree', 0.5,1,step=0.05\n        ),\n        'subsample': trial.suggest_float(\n            'subsample', 0.5,1,step=0.05\n        ),\n        'learning_rate': trial.suggest_float(\n            'learning_rate', 0.001,0.1,step=0.001\n        ),\n        'n_estimators': trial.suggest_int(\n            \"n_estimators\", 10,2000,10\n        ),\n        'max_depth': trial.suggest_int(\n            'max_depth', 2,20,1\n        ),\n        'random_state': 99,\n        'min_child_weight': trial.suggest_int(\n            'min_child_weight', 1,256,1\n        ),\n    }\n    \n    model = XGBClassifier(**param, enable_categorical = True) \n    \n    model.fit(X_train,y_train)\n    \n    preds = pd.DataFrame(model.predict(X_test))\n    \n    accuracy = sklearn.metrics.accuracy_score(pd.DataFrame(y_test.reset_index()['target']),preds)\n    \n    #accuracy = amex_metric_np(preds,pd.DataFrame(y_test.reset_index()['target']))\n    \n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:31:29.090942Z","iopub.execute_input":"2022-06-08T19:31:29.091319Z","iopub.status.idle":"2022-06-08T19:31:29.103956Z","shell.execute_reply.started":"2022-06-08T19:31:29.091286Z","shell.execute_reply":"2022-06-08T19:31:29.103034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials= 2)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:31:31.471725Z","iopub.execute_input":"2022-06-08T19:31:31.472357Z","iopub.status.idle":"2022-06-08T19:38:37.575754Z","shell.execute_reply.started":"2022-06-08T19:31:31.472319Z","shell.execute_reply":"2022-06-08T19:38:37.574758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = study.best_trial.params\nbest_params['tree_method'] = 'gpu_hist'\nbest_params['booster'] = 'gbtree'","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:39:21.781426Z","iopub.execute_input":"2022-06-08T19:39:21.781821Z","iopub.status.idle":"2022-06-08T19:39:21.787534Z","shell.execute_reply.started":"2022-06-08T19:39:21.78179Z","shell.execute_reply":"2022-06-08T19:39:21.786622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:39:38.575289Z","iopub.execute_input":"2022-06-08T19:39:38.575698Z","iopub.status.idle":"2022-06-08T19:39:38.582185Z","shell.execute_reply.started":"2022-06-08T19:39:38.575663Z","shell.execute_reply":"2022-06-08T19:39:38.581067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"best_params =  {'lambda': 2.8539273544398065,\n 'alpha': 4.539501914875077,\n 'colsample_bytree': 0.5,\n 'subsample': 1.0,\n 'learning_rate': 0.064,\n 'n_estimators': 950,\n 'max_depth': 14,\n 'min_child_weight': 27,\n 'tree_method': 'gpu_hist',\n 'booster': 'gbtree'}","metadata":{}},{"cell_type":"code","source":"final_model = XGBClassifier(**best_params,enable_categorical = True)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:39:38.594128Z","iopub.execute_input":"2022-06-08T19:39:38.595879Z","iopub.status.idle":"2022-06-08T19:39:38.600356Z","shell.execute_reply.started":"2022-06-08T19:39:38.595834Z","shell.execute_reply":"2022-06-08T19:39:38.599421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:39:38.616358Z","iopub.execute_input":"2022-06-08T19:39:38.617332Z","iopub.status.idle":"2022-06-08T19:43:04.208031Z","shell.execute_reply.started":"2022-06-08T19:39:38.6173Z","shell.execute_reply":"2022-06-08T19:43:04.20641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train,X_test,y_train,y_test\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:04.209936Z","iopub.execute_input":"2022-06-08T19:43:04.210347Z","iopub.status.idle":"2022-06-08T19:43:04.353214Z","shell.execute_reply.started":"2022-06-08T19:43:04.210306Z","shell.execute_reply":"2022-06-08T19:43:04.352194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_test_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(0) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading test data...')\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\ntest = read_test_file(path = TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:04.354786Z","iopub.execute_input":"2022-06-08T19:43:04.355869Z","iopub.status.idle":"2022-06-08T19:43:37.252467Z","shell.execute_reply.started":"2022-06-08T19:43:04.355814Z","shell.execute_reply":"2022-06-08T19:43:37.251551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:37.254815Z","iopub.execute_input":"2022-06-08T19:43:37.255215Z","iopub.status.idle":"2022-06-08T19:43:37.511236Z","shell.execute_reply.started":"2022-06-08T19:43:37.255174Z","shell.execute_reply":"2022-06-08T19:43:37.510106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = process_and_feature_engineer(test)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:37.512904Z","iopub.execute_input":"2022-06-08T19:43:37.513346Z","iopub.status.idle":"2022-06-08T19:43:45.345649Z","shell.execute_reply.started":"2022-06-08T19:43:37.513303Z","shell.execute_reply":"2022-06-08T19:43:45.344691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['prediction'] = final_model.predict_proba(test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:45.347109Z","iopub.execute_input":"2022-06-08T19:43:45.347712Z","iopub.status.idle":"2022-06-08T19:43:51.81117Z","shell.execute_reply.started":"2022-06-08T19:43:45.34767Z","shell.execute_reply":"2022-06-08T19:43:51.810131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = pd.DataFrame(test['prediction'].to_pandas())","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:51.812668Z","iopub.execute_input":"2022-06-08T19:43:51.813201Z","iopub.status.idle":"2022-06-08T19:43:52.195485Z","shell.execute_reply.started":"2022-06-08T19:43:51.813153Z","shell.execute_reply":"2022-06-08T19:43:52.194586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.to_csv(\"submission.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T19:43:52.196689Z","iopub.execute_input":"2022-06-08T19:43:52.197075Z","iopub.status.idle":"2022-06-08T19:43:56.577555Z","shell.execute_reply.started":"2022-06-08T19:43:52.197037Z","shell.execute_reply":"2022-06-08T19:43:56.576574Z"},"trusted":true},"execution_count":null,"outputs":[]}]}