{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# LOAD LIBRARIES\nimport os\nimport gc\nimport pickle\nfrom copy import deepcopy\nimport pandas as pd\nimport numpy as np # CPU libraries\nimport matplotlib.pyplot as plt\nimport cudf # GPU libraries\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nimport xgboost as xgb\nfrom catboost import Pool, CatBoostClassifier\nprint('RAPIDS version',cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:37:28.476438Z","iopub.execute_input":"2022-09-05T13:37:28.476960Z","iopub.status.idle":"2022-09-05T13:37:35.016270Z","shell.execute_reply.started":"2022-09-05T13:37:28.476860Z","shell.execute_reply":"2022-09-05T13:37:35.014519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 2\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5\n\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\n\nODIR = \"/kaggle/working/echesneau/\"\nif not os.path.isdir(ODIR):\n    os.makedirs(ODIR)\n\nTRAIN_SUBSAMPLE = 1.0\n\nresult_all = pd.DataFrame(columns=['model', 'preprocessing', 'name', \\\n                                   'y_valid_pred', 'y_pred','valid_acc', 'acc'])\nresult_sum = pd.DataFrame(columns=['model', 'preprocessing', 'name', 'y_pred', 'acc'])","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:37:37.836828Z","iopub.execute_input":"2022-09-05T13:37:37.838492Z","iopub.status.idle":"2022-09-05T13:37:37.867967Z","shell.execute_reply.started":"2022-09-05T13:37:37.838439Z","shell.execute_reply":"2022-09-05T13:37:37.866471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    \"\"\"\n    function to load dataset\n    The function is modified frm the original one\n    The Fillna is done only during the processing\n    \"\"\"\n    # LOAD DATAFRAME\n    if usecols is not None:\n        data = cudf.read_parquet(path, columns=usecols)\n    else:\n        data = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    data['customer_ID'] = data['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    data.S_2 = cudf.to_datetime( data.S_2 )\n    print('shape of data:', data.shape)\n\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:37:38.970847Z","iopub.execute_input":"2022-09-05T13:37:38.971296Z","iopub.status.idle":"2022-09-05T13:37:38.981627Z","shell.execute_reply.started":"2022-09-05T13:37:38.971262Z","shell.execute_reply":"2022-09-05T13:37:38.978470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(data):\n    \"\"\"\n    function to process database\n    FEATURE ENGINEERING FROM\n    https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    \"\"\"\n    all_cols = [c for c in list(data.columns) if c not in ['customer_ID','S_2']]\n    cat_feat = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\\\n                    \"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_feat]\n\n    test_num_agg = data.groupby(\"customer_ID\")[num_features].agg(['mean', \\\n                                                                  'std', \\\n                                                                  'min', \\\n                                                                  'max', \\\n                                                                  'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = data.groupby(\"customer_ID\")[cat_feat].agg(['count', \\\n                                                              'last', \\\n                                                              'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    data = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    data = data.fillna(NAN_VALUE)\n    print('shape after engineering', data.shape )\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:37:39.825670Z","iopub.execute_input":"2022-09-05T13:37:39.826153Z","iopub.status.idle":"2022-09-05T13:37:39.838411Z","shell.execute_reply.started":"2022-09-05T13:37:39.826121Z","shell.execute_reply":"2022-09-05T13:37:39.835882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric_mod(y_true, y_pred):\n    \"\"\"\n    function to calculate the metric of the competion\n    from https://www.kaggle.com/kyakovlev\n    and https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\n    \"\"\"\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:37:40.620511Z","iopub.execute_input":"2022-09-05T13:37:40.621320Z","iopub.status.idle":"2022-09-05T13:37:40.633550Z","shell.execute_reply.started":"2022-09-05T13:37:40.621287Z","shell.execute_reply":"2022-09-05T13:37:40.631329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# @huseyincot preprocessing","metadata":{}},{"cell_type":"markdown","source":"The processing proposed by @Huyseioncot seems to be interessting and it is one of the most used.\nSo we decide to base the predictions on this processing.","metadata":{}},{"cell_type":"markdown","source":"## Load and process","metadata":{}},{"cell_type":"markdown","source":"Parquet format is use to save GPU/RAM memory.","metadata":{}},{"cell_type":"code","source":"print('Reading train data...')\ntrain = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:37:44.317106Z","iopub.execute_input":"2022-09-05T13:37:44.317546Z","iopub.status.idle":"2022-09-05T13:38:11.527661Z","shell.execute_reply.started":"2022-09-05T13:37:44.317513Z","shell.execute_reply":"2022-09-05T13:38:11.526128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:11.531160Z","iopub.execute_input":"2022-09-05T13:38:11.531958Z","iopub.status.idle":"2022-09-05T13:38:14.838426Z","shell.execute_reply.started":"2022-09-05T13:38:11.531911Z","shell.execute_reply":"2022-09-05T13:38:14.836743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:14.841891Z","iopub.execute_input":"2022-09-05T13:38:14.843206Z","iopub.status.idle":"2022-09-05T13:38:15.783335Z","shell.execute_reply.started":"2022-09-05T13:38:14.843154Z","shell.execute_reply":"2022-09-05T13:38:15.780797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Targets are added in the database","metadata":{}},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:15.788151Z","iopub.execute_input":"2022-09-05T13:38:15.788685Z","iopub.status.idle":"2022-09-05T13:38:18.287083Z","shell.execute_reply.started":"2022-09-05T13:38:15.788637Z","shell.execute_reply":"2022-09-05T13:38:18.283385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(ODIR+'/all_features.pkl', 'wb') as ofile :\n    pickle.dump(FEATURES, ofile)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:18.290706Z","iopub.execute_input":"2022-09-05T13:38:18.291721Z","iopub.status.idle":"2022-09-05T13:38:18.300404Z","shell.execute_reply.started":"2022-09-05T13:38:18.291674Z","shell.execute_reply":"2022-09-05T13:38:18.298951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Features are needed for the prediction on the test set,\nwe save it.","metadata":{}},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"markdown","source":"XGBoost seems to be one of the most efficent model.","metadata":{}},{"cell_type":"code","source":"train = train.to_pandas() # free GPU memory\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:18.303014Z","iopub.execute_input":"2022-09-05T13:38:18.303745Z","iopub.status.idle":"2022-09-05T13:38:24.568610Z","shell.execute_reply.started":"2022-09-05T13:38:18.303701Z","shell.execute_reply":"2022-09-05T13:38:24.567068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\nxgb_parms = {\n    'max_depth':4,\n    'learning_rate':0.05,\n    'subsample':0.8,\n    'colsample_bytree':0.6,\n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:24.570965Z","iopub.execute_input":"2022-09-05T13:38:24.571774Z","iopub.status.idle":"2022-09-05T13:38:24.580872Z","shell.execute_reply.started":"2022-09-05T13:38:24.571729Z","shell.execute_reply":"2022-09-05T13:38:24.579109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Beause of memory limitation, database is split into flods.\nA model is train for each fold.\nAmex metric is calculated on the validation set, train set and all fold data.\nAt the end, the global metric is calculated","metadata":{}},{"cell_type":"code","source":"oof = []\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    dtrain = xgb.DMatrix(data=train.loc[train_idx, FEATURES], label=train.loc[train_idx, 'target'])\n    dvalid = xgb.DMatrix(data=train.loc[valid_idx, FEATURES], label=train.loc[valid_idx, 'target'])\n    model = xgb.train(xgb_parms,\n                      dtrain=dtrain,\n                      evals=[(dtrain,'train'),(dvalid,'valid')],\n                      num_boost_round=9999,\n                      #num_boost_round=99,\n                      early_stopping_rounds=100,\n                      verbose_eval=100)\n    model.save_model(f'{ODIR}/XGB_all_features_v{VER}_fold{fold}.xgb')\n    valid_pred = model.predict(dvalid)\n    val_acc = amex_metric_mod(train.loc[valid_idx, 'target'].values, valid_pred)\n    print('Kaggle Metric on valid set =',val_acc,'\\n')\n\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = valid_pred\n    oof.append( df )\n\n    del dtrain, dvalid, df\n    _ = gc.collect()\n\n    dall = xgb.DMatrix(data=train[FEATURES], label=train['target'])\n    pred = model.predict(dall)\n    all_acc = amex_metric_mod(train['target'].values, pred)\n    print('Kaggle Metric on all dataset =',all_acc,'\\n')\n    result_all = result_all.append({'model' : \"XGBoost\",\n                                    'preprocessing' : \"huseyincot_all_feat\",\n                                    'name' : f'XGB_all_features_v{VER}_fold{fold}',\n                                    'y_valid_pred' : valid_pred,\n                                    'valid_acc' : val_acc,\n                                    'y_pred' : pred,\n                                    'acc' : all_acc\n                                   },\n                                   ignore_index=True\n                                  )\n    del dall, pred, valid_pred\n    _ = gc.collect()\nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nresult_sum = result_sum.append({'model' : \"XGBoost\",\n                                'preprocessing':\"huseyincot_all_feat\",\n                                'name' : \"XGBoost_huseyincot_all_feat\",\n                                'y_pred' : oof,\n                                'acc': acc\n                               },\n                               ignore_index=True\n                              )\nconf_mat = confusion_matrix(oof.target.values, np.rint(oof.oof_pred.values), labels=[0,1],\n                           normalize='all')\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_mat,\n                              display_labels=[0,1])\ndisp.plot()\nplt.show()\nprint('OVERALL CV Kaggle Metric =',acc)\n\ndel oof, acc\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:38:24.583193Z","iopub.execute_input":"2022-09-05T13:38:24.584083Z","iopub.status.idle":"2022-09-05T13:50:29.424982Z","shell.execute_reply.started":"2022-09-05T13:38:24.584019Z","shell.execute_reply":"2022-09-05T13:50:29.423572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CatBoost","metadata":{}},{"cell_type":"markdown","source":"CatBoost is based on the same method than XGBoost but could  be more efficient.\nWe apply the same code than before but training a catboost.","metadata":{}},{"cell_type":"code","source":"# GET CATEG VARIABLES\ncat_features = [\"B_30\", \"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\ncateg = []\n#print(train.columns)\nfor col in train.columns :\n    if col not in ['customer_ID', 'target'] :\n        VAR = '_'.join(col.split('_')[:2])\n        if VAR in cat_features :\n            categ.append(col)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:50:29.426818Z","iopub.execute_input":"2022-09-05T13:50:29.428163Z","iopub.status.idle":"2022-09-05T13:50:29.438884Z","shell.execute_reply.started":"2022-09-05T13:50:29.428117Z","shell.execute_reply":"2022-09-05T13:50:29.437347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = []\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    train_pool = Pool(train.loc[train_idx, FEATURES],\n                      train.loc[train_idx, 'target'],\n                      cat_features=categ\n                     )\n    valid_pool = Pool(train.loc[valid_idx, FEATURES],\n                      train.loc[valid_idx, 'target'],\n                      cat_features=categ\n                     )\n    model = CatBoostClassifier(iterations=9999,\n                               random_state=SEED,\n                               task_type=\"GPU\",\n                               loss_function = 'Logloss',\n                               #learning_rate=0.05\n                               )\n    model.fit(train_pool, eval_set=valid_pool,\n              #od_type=\"Iter\",\n              early_stopping_rounds=100,\n              #od_wait=100,\n              verbose=100)\n    model.save_model(f'{ODIR}/CTB_all_features_v{VER}_fold{fold}.ctb')\n    valid_pred = model.predict_proba(valid_pool)[:,1]\n    val_acc = amex_metric_mod(train.loc[valid_idx, 'target'].values, valid_pred)\n    print('Kaggle Metric on valid set =',val_acc,'\\n')\n\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = valid_pred\n    oof.append( df )\n\n    del train_pool, valid_pool, df\n    _ = gc.collect()\n\n    all_pool = Pool(train[FEATURES],\n                    train['target'],\n                    cat_features=categ\n                     )\n    pred = model.predict_proba(all_pool)[:,1]\n    all_acc = amex_metric_mod(train['target'].values, pred)\n    print('Kaggle Metric on all dataset =',all_acc,'\\n')\n    result_all = result_all.append({'model' : \"CateBoost\",\n                                    'preprocessing' : \"huseyincot_all_feat\",\n                                    'name' : f'CTB_all_features_v{VER}_fold{fold}',\n                                    'y_valid_pred' : valid_pred,\n                                    'valid_acc' : val_acc,\n                                    'y_pred' : pred,\n                                    'acc' : all_acc\n                                   },\n                                   ignore_index=True\n                                  )\n    del all_pool, pred, valid_pred\n    _ = gc.collect()\n\nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nresult_sum = result_sum.append({'model' : \"CateBoost\",\n                                'preprocessing':\"huseyincot_all_feat\",\n                                'name' : \"CTB_huseyincot_all_feat\",\n                                'y_pred' : oof,\n                                'acc': acc\n                               },\n                               ignore_index=True\n                              )\nconf_mat = confusion_matrix(oof.target.values, np.rint(oof.oof_pred.values), labels=[0,1],\n                           normalize='all')\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_mat,\n                              display_labels=[0,1])\ndisp.plot()\nplt.show()\nprint('OVERALL CV Kaggle Metric =',acc)\n\ndel oof, acc\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T13:50:29.445631Z","iopub.execute_input":"2022-09-05T13:50:29.446110Z","iopub.status.idle":"2022-09-05T15:34:20.847204Z","shell.execute_reply.started":"2022-09-05T13:50:29.446053Z","shell.execute_reply":"2022-09-05T15:34:20.845717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\n_=gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:20.849183Z","iopub.execute_input":"2022-09-05T15:34:20.850346Z","iopub.status.idle":"2022-09-05T15:34:21.028843Z","shell.execute_reply.started":"2022-09-05T15:34:20.850287Z","shell.execute_reply":"2022-09-05T15:34:21.027097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Removing Columns with a majority of NaN","metadata":{}},{"cell_type":"markdown","source":"The EDA shows us that some features contain huge amount of NaN values.\nThese features are removed.","metadata":{}},{"cell_type":"markdown","source":"## Load Dataset","metadata":{}},{"cell_type":"code","source":"train = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:21.032534Z","iopub.execute_input":"2022-09-05T15:34:21.033421Z","iopub.status.idle":"2022-09-05T15:34:38.454124Z","shell.execute_reply.started":"2022-09-05T15:34:21.033374Z","shell.execute_reply":"2022-09-05T15:34:38.452568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Select features","metadata":{}},{"cell_type":"markdown","source":"Features are deleted if more than 20% of values are NaN.","metadata":{}},{"cell_type":"code","source":"counter = train.isnull().sum(axis=0).sort_values(ascending=False)/len(train)*100\nrm_nan = counter[counter>20].index\nrm_nan = list(rm_nan.to_array())\nprint(f\"{len(rm_nan)}/{len(train.columns)}\")","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:38.456529Z","iopub.execute_input":"2022-09-05T15:34:38.457521Z","iopub.status.idle":"2022-09-05T15:34:38.880296Z","shell.execute_reply.started":"2022-09-05T15:34:38.457471Z","shell.execute_reply":"2022-09-05T15:34:38.878775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES_2 = [col for col in train.columns if col not in rm_nan]","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:38.883503Z","iopub.execute_input":"2022-09-05T15:34:38.884522Z","iopub.status.idle":"2022-09-05T15:34:38.892850Z","shell.execute_reply.started":"2022-09-05T15:34:38.884473Z","shell.execute_reply":"2022-09-05T15:34:38.891131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[FEATURES_2]","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:38.895372Z","iopub.execute_input":"2022-09-05T15:34:38.896296Z","iopub.status.idle":"2022-09-05T15:34:38.930792Z","shell.execute_reply.started":"2022-09-05T15:34:38.896247Z","shell.execute_reply":"2022-09-05T15:34:38.929203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing","metadata":{}},{"cell_type":"markdown","source":"The same processing is applied","metadata":{}},{"cell_type":"code","source":"train = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:38.932988Z","iopub.execute_input":"2022-09-05T15:34:38.934214Z","iopub.status.idle":"2022-09-05T15:34:42.043866Z","shell.execute_reply.started":"2022-09-05T15:34:38.934154Z","shell.execute_reply":"2022-09-05T15:34:42.042485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:42.050033Z","iopub.execute_input":"2022-09-05T15:34:42.051381Z","iopub.status.idle":"2022-09-05T15:34:43.523409Z","shell.execute_reply.started":"2022-09-05T15:34:42.051331Z","shell.execute_reply":"2022-09-05T15:34:43.521841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES_2 = train.columns[1:-1]","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:43.526925Z","iopub.execute_input":"2022-09-05T15:34:43.528008Z","iopub.status.idle":"2022-09-05T15:34:43.537777Z","shell.execute_reply.started":"2022-09-05T15:34:43.527954Z","shell.execute_reply":"2022-09-05T15:34:43.534935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(ODIR+'/all_features_2.pkl', 'wb') as ofile :\n    pickle.dump(FEATURES_2, ofile)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:43.540697Z","iopub.execute_input":"2022-09-05T15:34:43.541862Z","iopub.status.idle":"2022-09-05T15:34:43.552763Z","shell.execute_reply.started":"2022-09-05T15:34:43.541801Z","shell.execute_reply":"2022-09-05T15:34:43.551238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"train = train.to_pandas() # free GPU memory\nTRAIN_SUBSAMPLE = 1.0\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:43.556377Z","iopub.execute_input":"2022-09-05T15:34:43.557930Z","iopub.status.idle":"2022-09-05T15:34:47.050824Z","shell.execute_reply.started":"2022-09-05T15:34:43.557883Z","shell.execute_reply":"2022-09-05T15:34:47.049522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\nxgb_parms = {\n    'max_depth':4,\n    'learning_rate':0.05,\n    'subsample':0.8,\n    'colsample_bytree':0.6,\n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:47.056357Z","iopub.execute_input":"2022-09-05T15:34:47.059708Z","iopub.status.idle":"2022-09-05T15:34:47.073912Z","shell.execute_reply.started":"2022-09-05T15:34:47.059659Z","shell.execute_reply":"2022-09-05T15:34:47.072028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = []\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    dtrain = xgb.DMatrix(data=train.loc[train_idx, FEATURES_2], \\\n                        label=train.loc[train_idx, 'target'])\n    dvalid = xgb.DMatrix(data=train.loc[valid_idx, FEATURES_2], \\\n                        label=train.loc[valid_idx, 'target'])\n    model = xgb.train(xgb_parms,\n                      dtrain=dtrain,\n                      evals=[(dtrain,'train'),(dvalid,'valid')],\n                      num_boost_round=9999,\n                      #num_boost_round=99,\n                      early_stopping_rounds=100,\n                      verbose_eval=100)\n    model.save_model(f'{ODIR}/XGB_nonan_features_v{VER}_fold{fold}.xgb')\n    valid_pred = model.predict(dvalid)\n    val_acc = amex_metric_mod(train.loc[valid_idx, 'target'].values, valid_pred)\n    print('Kaggle Metric on valid set =',val_acc,'\\n')\n\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = valid_pred\n    oof.append( df )\n\n    del dtrain, dvalid, df\n    _ = gc.collect()\n\n    dall = xgb.DMatrix(data=train[FEATURES_2], label=train['target'])\n    pred = model.predict(dall)\n    all_acc = amex_metric_mod(train['target'].values, pred)\n    print('Kaggle Metric on all dataset =',all_acc,'\\n')\n    #result_all = result_all.append({'model' : \"XGBoost\",\n    #                                'preprocessing' : \"huseyincot_nonan_feat\",\n    #                                'name' : f'XGB_nonan_features_v{VER}_fold{fold}',\n    #                                'y_valid_pred' : valid_pred,\n    #                                'valid_acc' : val_acc,\n    #                                'y_pred' : pred,\n    #                                'acc' : all_acc\n    #                               },\n    #                               ignore_index=True\n    #                              )\n    del dall, pred, valid_pred\n    _ = gc.collect()\nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\n#result_sum = result_sum.append({'model' : \"XGBoost\",\n#                                'preprocessing':\"huseyincot_nonan_feat\",\n#                                'name' : \"XGBoost_huseyincot_nonan_feat\",\n#                                'y_pred' : oof,\n#                                'acc': acc\n#                              },\n#                               ignore_index=True\n#                              )\nconf_mat = confusion_matrix(oof.target.values, np.rint(oof.oof_pred.values), labels=[0,1],\n                           normalize='all')\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_mat,\n                              display_labels=[0,1])\ndisp.plot()\nplt.show()\nprint('OVERALL CV Kaggle Metric =',acc)\n\ndel oof, acc\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:34:47.077664Z","iopub.execute_input":"2022-09-05T15:34:47.079562Z","iopub.status.idle":"2022-09-05T15:43:46.821136Z","shell.execute_reply.started":"2022-09-05T15:34:47.079522Z","shell.execute_reply":"2022-09-05T15:43:46.819609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CateBoost","metadata":{}},{"cell_type":"code","source":"# GET CATEG VARIABLES\ncat_features = [\"B_30\", \"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\ncateg = []\n#print(train.columns)\nfor col in train.columns :\n    if col not in ['customer_ID', 'target'] :\n        VAR = '_'.join(col.split('_')[:2])\n        if VAR in cat_features :\n            categ.append(col)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:43:46.824609Z","iopub.execute_input":"2022-09-05T15:43:46.825988Z","iopub.status.idle":"2022-09-05T15:43:46.836364Z","shell.execute_reply.started":"2022-09-05T15:43:46.825940Z","shell.execute_reply":"2022-09-05T15:43:46.835079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try :\n    del all_pool, pred, valid_pred\n    _ = gc.collect()\nexcept NameError :\n    pass","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:43:46.840186Z","iopub.execute_input":"2022-09-05T15:43:46.841539Z","iopub.status.idle":"2022-09-05T15:43:46.855661Z","shell.execute_reply.started":"2022-09-05T15:43:46.841490Z","shell.execute_reply":"2022-09-05T15:43:46.854091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = []\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    train_pool = Pool(train.loc[train_idx, FEATURES_2],\n                      train.loc[train_idx, 'target'],\n                      cat_features=categ\n                     )\n    valid_pool = Pool(train.loc[valid_idx, FEATURES_2],\n                      train.loc[valid_idx, 'target'],\n                      cat_features=categ\n                     )\n    model = CatBoostClassifier(iterations=9999,\n                               random_state=SEED,\n                               task_type=\"GPU\",\n                               loss_function = 'Logloss',\n                               #learning_rate=0.05\n                               )\n    model.fit(train_pool, eval_set=valid_pool,\n              #od_type=\"Iter\",\n              early_stopping_rounds=100,\n              #od_wait=100,\n              verbose=100)\n    model.save_model(f'{ODIR}/CTB_nonan_features_v{VER}_fold{fold}.ctb')\n    valid_pred = model.predict_proba(valid_pool)[:,1]\n    val_acc = amex_metric_mod(train.loc[valid_idx, 'target'].values, valid_pred)\n    print('Kaggle Metric on valid set =',val_acc,'\\n')\n\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = valid_pred\n    oof.append( df )\n\n    del train_pool, valid_pool, df\n    _ = gc.collect()\n\n    all_pool = Pool(train[FEATURES_2],\n                    train['target'],\n                    cat_features=categ\n                     )\n    pred = model.predict_proba(all_pool)[:,1]\n    all_acc = amex_metric_mod(train['target'].values, pred)\n    print('Kaggle Metric on all dataset =',all_acc,'\\n')\n    #result_all = result_all.append({'model' : \"CateBoost\",\n    #                                'preprocessing' : \"huseyincot_nonan_feat\",\n    #                                'name' : f'CTB_nonan_features_v{VER}_fold{fold}',\n    #                                'y_valid_pred' : valid_pred,\n    #                                'valid_acc' : val_acc\n    #                                'y_pred' : pred,\n    #                                'acc' : all_acc\n    #                               },\n    #                               ignore_index=True\n    #                              )\n    del all_pool, pred, valid_pred\n    _ = gc.collect()\n\nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\n#result_sum = result_sum.append({'model' : \"CateBoost\",\n#                                'preprocessing':\"huseyincot_all_feat\",\n#                                'name' : \"CTB_huseyincot_all_feat\",\n#                                'y_pred' : oof,\n#                                'acc': acc\n#                               },\n#                               ignore_index=True\n#                              )\nconf_mat = confusion_matrix(oof.target.values, np.rint(oof.oof_pred.values), labels=[0,1],\n                           normalize='all')\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_mat,\n                              display_labels=[0,1])\ndisp.plot()\nplt.show()\nprint('OVERALL CV Kaggle Metric =',acc)\n\ndel oof, acc\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T15:43:46.858683Z","iopub.execute_input":"2022-09-05T15:43:46.859379Z","iopub.status.idle":"2022-09-05T17:24:15.667209Z","shell.execute_reply.started":"2022-09-05T15:43:46.859328Z","shell.execute_reply":"2022-09-05T17:24:15.665592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\n_=gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:15.669683Z","iopub.execute_input":"2022-09-05T17:24:15.670585Z","iopub.status.idle":"2022-09-05T17:24:15.861663Z","shell.execute_reply.started":"2022-09-05T17:24:15.670524Z","shell.execute_reply":"2022-09-05T17:24:15.860110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features importances","metadata":{}},{"cell_type":"markdown","source":"A selection of most important features is done using the drop columns importances method.\nThe goal is to use only most important features for the prediction.","metadata":{}},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"train = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:15.863926Z","iopub.execute_input":"2022-09-05T17:24:15.864440Z","iopub.status.idle":"2022-09-05T17:24:39.426651Z","shell.execute_reply.started":"2022-09-05T17:24:15.864394Z","shell.execute_reply":"2022-09-05T17:24:39.425129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:39.435673Z","iopub.execute_input":"2022-09-05T17:24:39.436569Z","iopub.status.idle":"2022-09-05T17:24:42.979883Z","shell.execute_reply.started":"2022-09-05T17:24:39.436518Z","shell.execute_reply":"2022-09-05T17:24:42.978207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:42.983423Z","iopub.execute_input":"2022-09-05T17:24:42.984180Z","iopub.status.idle":"2022-09-05T17:24:44.771148Z","shell.execute_reply.started":"2022-09-05T17:24:42.984132Z","shell.execute_reply":"2022-09-05T17:24:44.769698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Because of the important size of the train set and in order to spped up the selection,\nwe select only 1/400 of rows.","metadata":{}},{"cell_type":"code","source":"train = train.loc[range(int(len(train)/400))]\ntrain=train.to_pandas()\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:44.775385Z","iopub.execute_input":"2022-09-05T17:24:44.777093Z","iopub.status.idle":"2022-09-05T17:24:45.622719Z","shell.execute_reply.started":"2022-09-05T17:24:44.777021Z","shell.execute_reply":"2022-09-05T17:24:45.618787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES_tmp = train.columns[1:-1]","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:45.629656Z","iopub.execute_input":"2022-09-05T17:24:45.630916Z","iopub.status.idle":"2022-09-05T17:24:45.651514Z","shell.execute_reply.started":"2022-09-05T17:24:45.630855Z","shell.execute_reply":"2022-09-05T17:24:45.649002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Select features","metadata":{}},{"cell_type":"code","source":"def dropcol_importances(rf_model, data, labels):\n    \"\"\"\n    Function to calculate features importances with a random forest model\n    \"\"\"\n    rf_model_ = deepcopy(rf_model)\n    rf_model_.random_state = 999\n    rf_model_.fit(data, labels)\n    baseline = rf_model_.oob_score_\n    imp = []\n    for i, column in enumerate(data.columns):\n        print(f\"{i}/{len(data.columns)}\", end=\"\\r\")\n        data_tmp = data.drop(column, axis=1)\n        rf_model_ = deepcopy(rf_model)\n        rf_model_.random_state = 999\n        rf_model_.fit(data_tmp, labels)\n        oob = rf_model_.oob_score_\n        imp.append(baseline - oob)\n    imp = np.array(imp)\n    out = pd.DataFrame(\n            data={'Feature':data.columns,\n                  'Importance':imp})\n    out = out.set_index('Feature')\n    out = out.sort_values('Importance', ascending=True)\n    return out","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:45.654985Z","iopub.execute_input":"2022-09-05T17:24:45.656482Z","iopub.status.idle":"2022-09-05T17:24:45.684930Z","shell.execute_reply.started":"2022-09-05T17:24:45.656435Z","shell.execute_reply":"2022-09-05T17:24:45.682977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestClassifier(\n         n_estimators=100,\n         # better generality with 5\n         min_samples_leaf=5,\n         n_jobs=-1,\n         oob_score=True)\nrf.fit(train[FEATURES_tmp], train['target']) # rf must be pre-trained","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:45.688570Z","iopub.execute_input":"2022-09-05T17:24:45.689079Z","iopub.status.idle":"2022-09-05T17:24:46.868528Z","shell.execute_reply.started":"2022-09-05T17:24:45.689005Z","shell.execute_reply":"2022-09-05T17:24:46.867060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dc_imp = dropcol_importances(rf,train[FEATURES_tmp] , train['target'])","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:24:46.870952Z","iopub.execute_input":"2022-09-05T17:24:46.871434Z","iopub.status.idle":"2022-09-05T17:39:13.282641Z","shell.execute_reply.started":"2022-09-05T17:24:46.871387Z","shell.execute_reply":"2022-09-05T17:39:13.281227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dc_imp","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:13.284763Z","iopub.execute_input":"2022-09-05T17:39:13.285250Z","iopub.status.idle":"2022-09-05T17:39:13.303807Z","shell.execute_reply.started":"2022-09-05T17:39:13.285205Z","shell.execute_reply":"2022-09-05T17:39:13.302148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The effect of each features on the accuracy is plot here","metadata":{}},{"cell_type":"code","source":"dc_imp.plot.barh()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:13.306580Z","iopub.execute_input":"2022-09-05T17:39:13.307695Z","iopub.status.idle":"2022-09-05T17:39:27.781683Z","shell.execute_reply.started":"2022-09-05T17:39:13.307639Z","shell.execute_reply":"2022-09-05T17:39:27.780113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only features with an importance > 0 are conserved","metadata":{}},{"cell_type":"code","source":"dc_imp[dc_imp[\"Importance\"]>0]","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:27.784154Z","iopub.execute_input":"2022-09-05T17:39:27.784694Z","iopub.status.idle":"2022-09-05T17:39:27.803969Z","shell.execute_reply.started":"2022-09-05T17:39:27.784633Z","shell.execute_reply":"2022-09-05T17:39:27.802312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES_3 = dc_imp[dc_imp[\"Importance\"]>0].index.to_list()\nwith open(ODIR+'/all_features_3.pkl', 'wb') as ofile :\n    pickle.dump(FEATURES_3, ofile)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:27.806493Z","iopub.execute_input":"2022-09-05T17:39:27.807192Z","iopub.status.idle":"2022-09-05T17:39:27.818871Z","shell.execute_reply.started":"2022-09-05T17:39:27.806971Z","shell.execute_reply":"2022-09-05T17:39:27.817332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PATH = \"../input/amex-output-echesneau\"\nif os.path.isfile(MODEL_PATH+\"/all_features_3.pkl\") :\n    with open(MODEL_PATH+\"/all_features_3.pkl\", 'rb') as f :\n        FEATURES_3 = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:27.821475Z","iopub.execute_input":"2022-09-05T17:39:27.822240Z","iopub.status.idle":"2022-09-05T17:39:27.871984Z","shell.execute_reply.started":"2022-09-05T17:39:27.822189Z","shell.execute_reply":"2022-09-05T17:39:27.870515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prepare data for modelization","metadata":{}},{"cell_type":"markdown","source":"The same processing is then applY.","metadata":{}},{"cell_type":"code","source":"train = read_file(path = TRAIN_PATH)\ntrain = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:27.873979Z","iopub.execute_input":"2022-09-05T17:39:27.875503Z","iopub.status.idle":"2022-09-05T17:39:33.582528Z","shell.execute_reply.started":"2022-09-05T17:39:27.875459Z","shell.execute_reply":"2022-09-05T17:39:33.580960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:33.586177Z","iopub.execute_input":"2022-09-05T17:39:33.587130Z","iopub.status.idle":"2022-09-05T17:39:35.276776Z","shell.execute_reply.started":"2022-09-05T17:39:33.587082Z","shell.execute_reply":"2022-09-05T17:39:35.275364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"train = train.to_pandas() # free GPU memory\nTRAIN_SUBSAMPLE = 1.0\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:35.280262Z","iopub.execute_input":"2022-09-05T17:39:35.281131Z","iopub.status.idle":"2022-09-05T17:39:39.606103Z","shell.execute_reply.started":"2022-09-05T17:39:35.281080Z","shell.execute_reply":"2022-09-05T17:39:39.604120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\nxgb_parms = {\n    'max_depth':4,\n    'learning_rate':0.05,\n    'subsample':0.8,\n    'colsample_bytree':0.6,\n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:39.608307Z","iopub.execute_input":"2022-09-05T17:39:39.609955Z","iopub.status.idle":"2022-09-05T17:39:39.620613Z","shell.execute_reply.started":"2022-09-05T17:39:39.609907Z","shell.execute_reply":"2022-09-05T17:39:39.618240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = []\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    dtrain = xgb.DMatrix(data=train.loc[train_idx, FEATURES_3], \\\n                        label=train.loc[train_idx, 'target'])\n    dvalid = xgb.DMatrix(data=train.loc[valid_idx, FEATURES_3], \\\n                         label=train.loc[valid_idx, 'target'])\n    model = xgb.train(xgb_parms,\n                      dtrain=dtrain,\n                      evals=[(dtrain,'train'),(dvalid,'valid')],\n                      num_boost_round=9999,\n                      #num_boost_round=99,\n                      early_stopping_rounds=100,\n                      verbose_eval=100)\n    model.save_model(f'{ODIR}/XGB_dc0_features_v{VER}_fold{fold}.xgb')\n    valid_pred = model.predict(dvalid)\n    val_acc = amex_metric_mod(train.loc[valid_idx, 'target'].values, valid_pred)\n    print('Kaggle Metric on valid set =',val_acc,'\\n')\n\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = valid_pred\n    oof.append( df )\n\n    del dtrain, dvalid, df\n    _ = gc.collect()\n\n    dall = xgb.DMatrix(data=train[FEATURES_3], label=train['target'])\n    pred = model.predict(dall)\n    all_acc = amex_metric_mod(train['target'].values, pred)\n    print('Kaggle Metric on all dataset =',all_acc,'\\n')\n    #result_all = result_all.append({'model' : \"XGBoost\",\n    #                                'preprocessing' : \"huseyincot_dc0_feat\",\n    #                                'name' : f'XGB_dc0_features_v{VER}_fold{fold}',\n    #                                'y_valid_pred' : valid_pred,\n    #                                'valid_acc' : val_acc,\n    #                                'y_pred' : pred,\n    #                                'acc' : all_acc\n    #                               },\n    #                               ignore_index=True\n    #                              )\n    del dall, pred, valid_pred\n    _ = gc.collect()\nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\n#result_sum = result_sum.append({'model' : \"XGBoost\",\n#                                'preprocessing':\"huseyincot_dc0_feat\",\n#                                'name' : \"XGBoost_huseyincot_dc0_feat\",\n#                                'y_pred' : oof,\n#                                'acc': acc\n#                              },\n#                               ignore_index=True\n#                              )\nconf_mat = confusion_matrix(oof.target.values, np.rint(oof.oof_pred.values), labels=[0,1],\n                           normalize='all')\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_mat,\n                              display_labels=[0,1])\ndisp.plot()\nplt.show()\nprint('OVERALL CV Kaggle Metric =',acc)\n\ndel oof, acc\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:39:39.622929Z","iopub.execute_input":"2022-09-05T17:39:39.623829Z","iopub.status.idle":"2022-09-05T17:46:13.599399Z","shell.execute_reply.started":"2022-09-05T17:39:39.623765Z","shell.execute_reply":"2022-09-05T17:46:13.597892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CateBoost","metadata":{}},{"cell_type":"code","source":"# GET CATEG VARIABLES\ncat_features = [\"B_30\", \"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\ncateg = []\n#print(train.columns)\nfor col in FEATURES_3 :\n    if col not in ['customer_ID', 'target'] :\n        VAR = '_'.join(col.split('_')[:2])\n        if VAR in cat_features :\n            categ.append(col)","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:46:13.601038Z","iopub.execute_input":"2022-09-05T17:46:13.601874Z","iopub.status.idle":"2022-09-05T17:46:13.610060Z","shell.execute_reply.started":"2022-09-05T17:46:13.601840Z","shell.execute_reply":"2022-09-05T17:46:13.608218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = []\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    train_pool = Pool(train.loc[train_idx, FEATURES_3],\n                      train.loc[train_idx, 'target'],\n                      cat_features=categ\n                     )\n    valid_pool = Pool(train.loc[valid_idx, FEATURES_3],\n                      train.loc[valid_idx, 'target'],\n                      cat_features=categ\n                     )\n    model = CatBoostClassifier(iterations=9999,\n                               random_state=SEED,\n                               task_type=\"GPU\",\n                               loss_function = 'Logloss',\n                               #learning_rate=0.05\n                               )\n    model.fit(train_pool, eval_set=valid_pool,\n              #od_type=\"Iter\",\n              early_stopping_rounds=100,\n              #od_wait=100,\n              verbose=100)\n    model.save_model(f'{ODIR}/CTB_dc0_features_v{VER}_fold{fold}.ctb')\n    valid_pred = model.predict_proba(valid_pool)[:,1]\n    val_acc = amex_metric_mod(train.loc[valid_idx, 'target'].values, valid_pred)\n    print('Kaggle Metric on valid set =',val_acc,'\\n')\n\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = valid_pred\n    oof.append( df )\n\n    del train_pool, valid_pool, df\n    _ = gc.collect()\n\n    all_pool = Pool(train[FEATURES_3],\n                    train['target'],\n                    cat_features=categ\n                     )\n    pred = model.predict_proba(all_pool)[:,1]\n    all_acc = amex_metric_mod(train['target'].values, pred)\n    print('Kaggle Metric on all dataset =',all_acc,'\\n')\n    #result_all = result_all.append({'model' : \"CateBoost\",\n    #                                'preprocessing' : \"huseyincot_dc0_feat\",\n    #                                'name' : f'CTB_dc0_features_v{VER}_fold{fold}',\n    #                                'y_valid_pred' : valid_pred,\n    #                                'valid_acc' : val_acc,\n    #                                'y_pred' : pred,\n    #                                'acc' : all_acc\n    #                               },\n    #                               ignore_index=True\n    #                              )\n    del all_pool, pred, valid_pred\n    _ = gc.collect()\n\nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\n#result_sum = result_sum.append({'model' : \"CateBoost\",\n#                                'preprocessing':\"huseyincot_all_feat\",\n#                                'name' : \"CTB_huseyincot_all_feat\",\n#                                'y_pred' : oof,\n#                                'acc': acc\n#                               },\n#                               ignore_index=True\n#                              )\nconf_mat = confusion_matrix(oof.target.values, np.rint(oof.oof_pred.values), labels=[0,1],\n                           normalize='all')\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_mat,\n                              display_labels=[0,1])\ndisp.plot()\nplt.show()\nprint('OVERALL CV Kaggle Metric =',acc)\n\ndel oof, acc\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T17:46:13.612802Z","iopub.execute_input":"2022-09-05T17:46:13.613842Z","iopub.status.idle":"2022-09-05T18:22:48.049406Z","shell.execute_reply.started":"2022-09-05T17:46:13.613798Z","shell.execute_reply":"2022-09-05T18:22:48.047801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\n_=gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-05T18:22:48.052509Z","iopub.execute_input":"2022-09-05T18:22:48.053481Z","iopub.status.idle":"2022-09-05T18:22:48.350972Z","shell.execute_reply.started":"2022-09-05T18:22:48.053437Z","shell.execute_reply":"2022-09-05T18:22:48.349322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All Model should be download in order to be used in \nanother notebook to create the submission file.","metadata":{}},{"cell_type":"code","source":"from IPython.display import FileLink \nimport os\n!pwd\n%cd /kaggle/working/echesneau\n!zip -r my_model.zip *","metadata":{"execution":{"iopub.status.busy":"2022-09-05T18:22:48.354588Z","iopub.execute_input":"2022-09-05T18:22:48.355621Z","iopub.status.idle":"2022-09-05T18:23:11.036655Z","shell.execute_reply.started":"2022-09-05T18:22:48.355571Z","shell.execute_reply":"2022-09-05T18:23:11.034897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-09-05T18:27:28.817283Z","iopub.execute_input":"2022-09-05T18:27:28.818442Z","iopub.status.idle":"2022-09-05T18:27:30.546024Z","shell.execute_reply.started":"2022-09-05T18:27:28.818394Z","shell.execute_reply":"2022-09-05T18:27:30.544232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a href=\"echesneau/my_model.zip\"> Download File </a>","metadata":{}}]}