{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# LGBM Optuna Starter\n\nThis notebook shows how to use [Optuna][1] in LGBM Models to automate hyper parameter searching.\n\n[1]: https://optuna.org/\n\n\nReferences\n\n(1) https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n\n(2) https://www.kaggle.com/code/cdeotte/xgboost-starter-0-793\n\n(3) https://www.kaggle.com/competitions/amex-default-prediction/discussion/328606","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np # CPU libraries\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\n\nprint('RAPIDS version',cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:51:28.579304Z","iopub.execute_input":"2022-07-30T01:51:28.579878Z","iopub.status.idle":"2022-07-30T01:51:33.018706Z","shell.execute_reply.started":"2022-07-30T01:51:28.579772Z","shell.execute_reply":"2022-07-30T01:51:33.017835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 1\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:51:39.661307Z","iopub.execute_input":"2022-07-30T01:51:39.661882Z","iopub.status.idle":"2022-07-30T01:51:39.666574Z","shell.execute_reply.started":"2022-07-30T01:51:39.661847Z","shell.execute_reply":"2022-07-30T01:51:39.665536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process and Feature Engineer Train Data\nI will use the features and data preprocessing methods introduced in the XGBoost starter notebook(2)\n\n\nWe will load @raddar Kaggle dataset from [here][1] with discussion [here][2]. Then we will engineer features suggested by @huseyincot in his notebooks [here][3] and [here][4]. We will use [RAPIDS][5] and the GPU to create new features quickly.\n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514\n[3]: https://www.kaggle.com/code/huseyincot/amex-catboost-0-793\n[4]: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n[5]: https://rapids.ai/","metadata":{}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:51:40.595471Z","iopub.execute_input":"2022-07-30T01:51:40.596031Z","iopub.status.idle":"2022-07-30T01:51:40.601867Z","shell.execute_reply.started":"2022-07-30T01:51:40.595995Z","shell.execute_reply":"2022-07-30T01:51:40.601028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:51:41.021772Z","iopub.execute_input":"2022-07-30T01:51:41.022487Z","iopub.status.idle":"2022-07-30T01:51:41.031073Z","shell.execute_reply.started":"2022-07-30T01:51:41.022450Z","shell.execute_reply":"2022-07-30T01:51:41.030205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = TRAIN_PATH)\n\ntrain = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:51:41.474195Z","iopub.execute_input":"2022-07-30T01:51:41.474785Z","iopub.status.idle":"2022-07-30T01:52:06.208156Z","shell.execute_reply.started":"2022-07-30T01:51:41.474748Z","shell.execute_reply":"2022-07-30T01:52:06.207312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets.index = targets['customer_ID'].sort_index()\ntargets = targets.drop('customer_ID', axis=1)\ntrain = train.join(targets,on =['customer_ID'] ).sort_index()\n\ndel targets\ngc.collect()\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:06.210053Z","iopub.execute_input":"2022-07-30T01:52:06.210619Z","iopub.status.idle":"2022-07-30T01:52:07.826806Z","shell.execute_reply.started":"2022-07-30T01:52:06.210580Z","shell.execute_reply":"2022-07-30T01:52:07.825994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Faster metric Implementation\n\n[Reference][2]\n\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n\ndef lgb_amex_metric(y_true, y_pred):\n    return ('Score',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:07.828071Z","iopub.execute_input":"2022-07-30T01:52:07.828499Z","iopub.status.idle":"2022-07-30T01:52:07.837631Z","shell.execute_reply.started":"2022-07-30T01:52:07.828460Z","shell.execute_reply":"2022-07-30T01:52:07.836637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nimport warnings\nimport gc\nimport pickle\n\nimport sklearn\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:07.839870Z","iopub.execute_input":"2022-07-30T01:52:07.840640Z","iopub.status.idle":"2022-07-30T01:52:08.564716Z","shell.execute_reply.started":"2022-07-30T01:52:07.840604Z","shell.execute_reply":"2022-07-30T01:52:08.563766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Use Optuna For HyperParameter Tuning","metadata":{}},{"cell_type":"code","source":"import optuna","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:08.568243Z","iopub.execute_input":"2022-07-30T01:52:08.568553Z","iopub.status.idle":"2022-07-30T01:52:08.572750Z","shell.execute_reply.started":"2022-07-30T01:52:08.568520Z","shell.execute_reply":"2022-07-30T01:52:08.571575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = train.to_pandas()\ndel train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:08.574513Z","iopub.execute_input":"2022-07-30T01:52:08.575018Z","iopub.status.idle":"2022-07-30T01:52:12.907328Z","shell.execute_reply.started":"2022-07-30T01:52:08.574958Z","shell.execute_reply":"2022-07-30T01:52:12.906420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, test_df = train_test_split(train_pd, test_size=0.25, stratify=train_pd['target'])\ndel train_pd\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:12.908616Z","iopub.execute_input":"2022-07-30T01:52:12.909046Z","iopub.status.idle":"2022-07-30T01:52:15.300324Z","shell.execute_reply.started":"2022-07-30T01:52:12.909008Z","shell.execute_reply":"2022-07-30T01:52:15.299421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_df.drop(['customer_ID', 'target'], axis=1)\nX_test = test_df.drop(['customer_ID', 'target'], axis=1)\ny_train = train_df['target']\ny_test = test_df['target']\n\ndel train_df, test_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:52:15.301832Z","iopub.execute_input":"2022-07-30T01:52:15.302210Z","iopub.status.idle":"2022-07-30T01:52:16.219430Z","shell.execute_reply.started":"2022-07-30T01:52:15.302175Z","shell.execute_reply":"2022-07-30T01:52:16.218706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Define an objective function to be maximized.\ndef objective(trial):\n    dtrain = lgb.Dataset( X_train, label=y_train)\n    # 2. Suggest values of the hyperparameters using a trial object.\n    param = {\n        'objective': 'binary',\n        'metric': 'binary_logloss',\n        'seed' : 42,\n        'lambda_l1': trial.suggest_float('lambda_l1', 1e-8, 10.0, log=True),\n        'lambda_l2': trial.suggest_float('lambda_l2', 1e-8, 10.0, log=True),\n        'num_leaves': trial.suggest_int('num_leaves', 2, 256),\n        'feature_fraction': trial.suggest_float('feature_fraction', 0.1, 1.0),\n        'bagging_fraction': trial.suggest_float('bagging_fraction', 0.1, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n        'min_data_in_leaf': trial.suggest_int('min_child_samples', 5, 100),\n        'learning_rate': trial.suggest_float('learning_rate', 0.001, 0.05, step=0.001),\n        'device' : 'gpu',\n        \"verbosity\": -1,\n    }\n\n    gbm = lgb.train(param, dtrain)\n    preds = gbm.predict(X_test)\n    pred_labels = np.rint(preds)\n    accuracy = sklearn.metrics.accuracy_score(y_test, pred_labels)\n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:22:26.092976Z","iopub.execute_input":"2022-07-30T02:22:26.093703Z","iopub.status.idle":"2022-07-30T02:22:26.103002Z","shell.execute_reply.started":"2022-07-30T02:22:26.093667Z","shell.execute_reply":"2022-07-30T02:22:26.102137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# 3. Create a study object and optimize the objective function.\n\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=100)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:22:31.830760Z","iopub.execute_input":"2022-07-30T02:22:31.831115Z","iopub.status.idle":"2022-07-30T04:56:45.692223Z","shell.execute_reply.started":"2022-07-30T02:22:31.831086Z","shell.execute_reply":"2022-07-30T04:56:45.691417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trial = study.best_trial\ntrial.params","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:02:22.628113Z","iopub.execute_input":"2022-07-30T05:02:22.628493Z","iopub.status.idle":"2022-07-30T05:02:22.634319Z","shell.execute_reply.started":"2022-07-30T05:02:22.628462Z","shell.execute_reply":"2022-07-30T05:02:22.633576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Result Visualization\n\n[Reference][3]\n\n[3]:[https://optuna.readthedocs.io/en/stable/reference/visualization/index.html]","metadata":{}},{"cell_type":"code","source":"optuna.visualization.plot_parallel_coordinate(study)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:03:13.747828Z","iopub.execute_input":"2022-07-30T05:03:13.748188Z","iopub.status.idle":"2022-07-30T05:03:13.922652Z","shell.execute_reply.started":"2022-07-30T05:03:13.748157Z","shell.execute_reply":"2022-07-30T05:03:13.921886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_slice(study)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:08:27.319735Z","iopub.execute_input":"2022-07-30T05:08:27.320550Z","iopub.status.idle":"2022-07-30T05:08:27.519666Z","shell.execute_reply.started":"2022-07-30T05:08:27.320508Z","shell.execute_reply":"2022-07-30T05:08:27.518929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_optimization_history(study)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:07:33.321228Z","iopub.execute_input":"2022-07-30T05:07:33.322162Z","iopub.status.idle":"2022-07-30T05:07:33.359926Z","shell.execute_reply.started":"2022-07-30T05:07:33.322123Z","shell.execute_reply":"2022-07-30T05:07:33.358946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(study)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:05:31.495144Z","iopub.execute_input":"2022-07-30T05:05:31.495785Z","iopub.status.idle":"2022-07-30T05:05:34.456105Z","shell.execute_reply.started":"2022-07-30T05:05:31.495748Z","shell.execute_reply":"2022-07-30T05:05:34.455389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train LightGBM","metadata":{}},{"cell_type":"code","source":"final_model = lgb.LGBMClassifier(**trial.params)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:11:14.638432Z","iopub.execute_input":"2022-07-30T05:11:14.639252Z","iopub.status.idle":"2022-07-30T05:11:14.643442Z","shell.execute_reply.started":"2022-07-30T05:11:14.639217Z","shell.execute_reply":"2022-07-30T05:11:14.642300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_m","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [f for f in train.columns if f != 'customer_ID' and f != 'target']\nprint(\"Number of Features :\",len(features))\ndef lgbm_params(random_state=1, n_estimators=1200):\n    return LGBMClassifier(n_estimators=n_estimators,\n                          #boosting_type = 'dart',\n                          learning_rate=0.03, reg_lambda=50,\n                          min_child_samples=2400,\n                          num_leaves=95,\n                          colsample_bytree=0.19,\n                          device='gpu',\n                          random_state=random_state)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:52:08.834994Z","iopub.execute_input":"2022-07-17T12:52:08.835307Z","iopub.status.idle":"2022-07-17T12:52:08.845629Z","shell.execute_reply.started":"2022-07-17T12:52:08.835276Z","shell.execute_reply":"2022-07-17T12:52:08.841436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_list = []\nkf = StratifiedKFold(n_splits=5)\n\nfor fold, (train_idx, valid_idx) in enumerate(kf.split(train, train.target.to_array())):\n    print('#'*25)\n    print('### Fold',fold)\n    print('### Train size:',len(train_idx),', Validation size:',len(valid_idx))\n    print('#'*25)\n    X_tr, X_val, y_tr, y_val, model = None, None, None, None, None\n    start_time = datetime.datetime.now()\n    X_tr = train.iloc[train_idx][features].as_gpu_matrix()\n    X_val = train.iloc[valid_idx][features].as_gpu_matrix()\n    y_tr = cupy.asarray(train.iloc[train_idx][\"target\"])\n    y_val = cupy.asarray(train.iloc[valid_idx][\"target\"])\n    \n    model = lgbm_params()\n    with warnings.catch_warnings():\n        warnings.filterwarnings('ignore', category=UserWarning)\n        model.fit(X_tr, cupy.asnumpy(y_tr),\n                  eval_set = [(X_val, cupy.asnumpy(y_val))], \n                  eval_metric=[lgb_amex_metric],\n                  callbacks=[log_evaluation(100)])\n        file = f'LGBM_v{VER}_fold{fold}.pkl'\n        pickle.dump(model, open(file, 'wb'))\n    y_val_pred = model.predict_proba(X_val, raw_score=True)\n    score = amex_metric(y_val, y_val_pred)\n    n_trees = model.best_iteration_\n    if n_trees is None: n_trees = model.n_estimators\n    print(f\"\\n\\n\\nFold {fold} | Training Time: {str(datetime.datetime.now() - start_time)[-12:-7]} |\"\n          f\" Score = {score:.5f}\\n\\n\\n\")\n    score_list.append(score)\n    \n    del X_val, y_val, score, model\n    gc.collect()\n    \nprint(f\"\\n\\n\\nScore: {np.mean(cupy.asarray(score_list)):.5f}\\n\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:52:40.550322Z","iopub.execute_input":"2022-07-17T12:52:40.550712Z","iopub.status.idle":"2022-07-17T13:20:55.904312Z","shell.execute_reply.started":"2022-07-17T12:52:40.550682Z","shell.execute_reply":"2022-07-17T13:20:55.903479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:20:55.905999Z","iopub.execute_input":"2022-07-17T13:20:55.906337Z","iopub.status.idle":"2022-07-17T13:20:56.165731Z","shell.execute_reply.started":"2022-07-17T13:20:55.906302Z","shell.execute_reply":"2022-07-17T13:20:56.164398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Predictions\nSince the Test Data is big, predicting all the results at once leads to an memory error. Split the data into 4 parts, make each prediction, and append them.","metadata":{}},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4\nTEST_PATH =  '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:20:56.167288Z","iopub.execute_input":"2022-07-17T13:20:56.168052Z","iopub.status.idle":"2022-07-17T13:20:58.776757Z","shell.execute_reply.started":"2022-07-17T13:20:56.16801Z","shell.execute_reply":"2022-07-17T13:20:58.775791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    \n    # READ PART OF TEST DATA\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n    \n    # PROCESS AND FEATURE ENGINEER PART OF TEST DATA\n    test = process_and_feature_engineer(test)\n    test = test.fillna(NAN_VALUE)\n    if k==NUM_PARTS-1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    dtest = test[features].as_gpu_matrix()\n    del test \n    gc.collect()\n    # reduce memory test = test[['P_2_mean']] 必要だったら追加\n     # INFER LGBM MODELS ON TEST DATA\n    with open(f'LGBM_v{VER}_fold0.pkl', 'rb') as pickle_file:\n        model = pickle.load(pickle_file)\n        preds = model.predict_proba(dtest,raw_score=True)\n    for f in range(1,FOLDS):\n        with open(f'LGBM_v{VER}_fold{f}.pkl', 'rb') as pickle_file:\n            model = pickle.load(pickle_file)\n            preds += model.predict_proba(dtest,raw_score=True)\n    preds = preds / FOLDS\n    test_preds.append(preds)\n\n# CLEAN MEMORY\ndel dtest, model\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:20:58.778939Z","iopub.execute_input":"2022-07-17T13:20:58.7793Z","iopub.status.idle":"2022-07-17T13:34:59.311963Z","shell.execute_reply.started":"2022-07-17T13:20:58.779264Z","shell.execute_reply":"2022-07-17T13:34:59.311122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make submission","metadata":{}},{"cell_type":"code","source":"test_predictions = np.concatenate(test_preds)\n\nsubmission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsubmission.loc[:, \"prediction\"] = test_predictions\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:34:59.313234Z","iopub.execute_input":"2022-07-17T13:34:59.313583Z","iopub.status.idle":"2022-07-17T13:35:05.496647Z","shell.execute_reply.started":"2022-07-17T13:34:59.31355Z","shell.execute_reply":"2022-07-17T13:35:05.495883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:35:05.497839Z","iopub.execute_input":"2022-07-17T13:35:05.4984Z","iopub.status.idle":"2022-07-17T13:35:05.513215Z","shell.execute_reply.started":"2022-07-17T13:35:05.498362Z","shell.execute_reply":"2022-07-17T13:35:05.512454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Future Ideas\n\n- Use Optuna for hyperparmaeter tuning\n- Change Boosting method to DART: slower but better accuracy","metadata":{}}]}