{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nimport optuna\nfrom functools import partial\nimport gc; gc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-16T17:43:46.458638Z","iopub.execute_input":"2022-06-16T17:43:46.459289Z","iopub.status.idle":"2022-06-16T17:43:46.466006Z","shell.execute_reply.started":"2022-06-16T17:43:46.459242Z","shell.execute_reply":"2022-06-16T17:43:46.465291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1 = pd.read_csv('../input/avg-weights/submission.csv')\ntraini = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', parse_dates=['S_2'], chunksize=450_000, iterator=True)\ntesti = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', parse_dates=['S_2'], chunksize=400_000, iterator=True) \nlabels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:43:46.46725Z","iopub.execute_input":"2022-06-16T17:43:46.467661Z","iopub.status.idle":"2022-06-16T17:43:49.919252Z","shell.execute_reply.started":"2022-06-16T17:43:46.467613Z","shell.execute_reply":"2022-06-16T17:43:49.91838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = []\nfor df in traini:\n    if len(train)>0: train = pd.concat([train, df])\n    else: train = df[:]\n    train.sort_values(by=['S_2'], inplace=True)\n    train.reset_index(drop=True, inplace=True)\n    train.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()\ntrain = pd.merge(train, labels, how='inner', on=['customer_ID'])\ndel labels; gc.collect()\ncol = [c for c in train if c not in ['customer_ID', 'target','S_2']]\ntrain.fillna(0).to_csv('train.csv', index=False)\ndel train; del traini; gc.collect()\n\ntest = []\nfor df in testi:\n    if len(test)>0: test = pd.concat([test, df])\n    else: test = df[:]\n    test.sort_values(by=['S_2'], inplace=True)\n    test.reset_index(drop=True, inplace=True)\n    test.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()\ntest.fillna(0).to_csv('test.csv', index=False)\ndel test; del testi; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T16:43:20.631243Z","iopub.execute_input":"2022-06-16T16:43:20.631745Z","iopub.status.idle":"2022-06-16T17:14:19.331768Z","shell.execute_reply.started":"2022-06-16T16:43:20.631694Z","shell.execute_reply":"2022-06-16T17:14:19.3309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:14:19.33402Z","iopub.execute_input":"2022-06-16T17:14:19.334403Z","iopub.status.idle":"2022-06-16T17:14:19.350711Z","shell.execute_reply.started":"2022-06-16T17:14:19.334369Z","shell.execute_reply":"2022-06-16T17:14:19.349754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('train.csv')\ncat_features = ['B_30', 'B_31', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nfor c in cat_features: train[c] = train[c].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:14:19.351762Z","iopub.execute_input":"2022-06-16T17:14:19.352929Z","iopub.status.idle":"2022-06-16T17:14:52.011268Z","shell.execute_reply.started":"2022-06-16T17:14:19.352878Z","shell.execute_reply":"2022-06-16T17:14:52.009753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train[col]\ny = train.target","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:14:53.418961Z","iopub.execute_input":"2022-06-16T17:14:53.419336Z","iopub.status.idle":"2022-06-16T17:14:54.021139Z","shell.execute_reply.started":"2022-06-16T17:14:53.419301Z","shell.execute_reply":"2022-06-16T17:14:54.02017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def optimize(trial, x, y):\n    \n    param_grid = {\n        #'task_type': 'GPU',\n        #'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-3, 10.0),\n        'max_bin': trial.suggest_int('max_bin', 200, 400),\n        #'subsample': trial.suggest_uniform('bagging_fraction', 0.4, 1.0),\n        'learning_rate': trial.suggest_uniform('learning_rate', 0.006, 0.018),\n        'n_estimators':  trial.suggest_int('n_estimators',1000,10000),\n        #'iterations': trial.suggest_int('iterations',500,2000),\n        'max_depth': trial.suggest_categorical('max_depth', [5,7,9,11,13,15]),\n        'random_state': trial.suggest_int('random_state', 22,200),\n        #'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 150),\n        'verbose': False,\n        'early_stopping_rounds': trial.suggest_categorical('early_stopping_rounds', [0,5,10,15,20]),\n        'bootstrap_type':'Poisson'\n        #'random_seed': trial.suggest_int('iterations',0,100),\n    }\n    \n    train_x, test_x, train_y, test_y = train_test_split(x, y, test_size=0.25, random_state=42)\n    \n    model = CatBoostClassifier(**param_grid)  \n    \n    model.fit(train_x,train_y,eval_set=[(test_x,test_y)], cat_features=cat_features)\n    \n    preds = model.predict(test_x)[:,1]\n    \n    amex_score = amex_metric(test_y, preds)\n    \n    return amex_score","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:26:16.78934Z","iopub.execute_input":"2022-06-16T17:26:16.789856Z","iopub.status.idle":"2022-06-16T17:26:16.801967Z","shell.execute_reply.started":"2022-06-16T17:26:16.789815Z","shell.execute_reply":"2022-06-16T17:26:16.800421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\noptimizer_function = partial(optimize, x=X, y=y)\n\nstudy = optuna.create_study(direction='maximize', sampler=optuna.samplers.TPESampler(), study_name=\"CatBoostClassifier\")\nstudy.optimize(optimizer_function, n_trials=50)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T17:26:19.201081Z","iopub.execute_input":"2022-06-16T17:26:19.201512Z","iopub.status.idle":"2022-06-16T17:43:46.323202Z","shell.execute_reply.started":"2022-06-16T17:26:19.201478Z","shell.execute_reply":"2022-06-16T17:43:46.322152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = study.best_trial.params\nprint(study.best_trial.params)\n_ = gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1, x2, y1, y2 = train_test_split(X, y, test_size=0.20, random_state=42)\ndel X,y\n_ = gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = CatBoostClassifier(**best_params)\nclf.fit(x1, y1, eval_set=[(x2, y2)], cat_features=cat_features)\n# preds = clf.predict_proba(x2)[:, 1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x1,x2,y1,y2\n_ = gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('test.csv')\nfor c in cat_features: test[c] = test[c].astype(str)\ntest['prediction'] = clf.predict_proba(test[col])[:, 1]\nsub2 = test[['customer_ID', 'prediction']]\ndel test;  gc.collect()\nos.remove ('test.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub2.columns = ['customer_ID', 'prediction2']\nblend = pd.merge(sub1, sub2, how='inner', on='customer_ID')\nblend.prediction = (blend.prediction * 0.955 + blend.prediction2 * 0.045)\nblend[['customer_ID', 'prediction']].to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### With K-Folds","metadata":{}},{"cell_type":"code","source":"# NFolds = 3\n# skf = StratifiedKFold(n_splits=NFolds,shuffle=True,random_state=42)\n# y_oof = np.zeros(train_x.shape[0])\n# y_test = np.zeros(test.shape[0])\n# ix=0\n# for trn_idx, test_idx in kf.split(X,y):\n#     print(f\"******* Fold {ix} ******* \")\n#     tr_x,val_x = (X.iloc[train_ind].reset_index(drop=True),X.iloc[val_ind].reset_index(drop=True))\n#     tr_y,val_y = (y.iloc[train_ind].reset_index(drop=True),y.iloc[val_ind].reset_index(drop=True))    \n    \n#     clf = CatBoostClassifier(**best_params)\n    \n#     clf.fit(tr_x,tr_y, eval_set=[(val_x,val_y)],cat_features=cat_features, verbose=100)\n#     preds = clf.predict_proba(val_x)[:, 1]\n#     y_oof[val_ind] = y_oof[val_ind] + preds\n#     preds_test = clf.predict_proba(test)[:, 1]\n#     y_test = y_test + preds_test / NFolds\n#     ix = ix + 1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = train_y.copy(deep=True)\n# y_pred = y_pred.rename(columns={\"target\": \"prediction\"})\n# y_pred[\"prediction\"] = y_oof\n# val_score = amex_metric(train_y, y_pred)\n# print(f\"Amex metric: {val_score}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test[\"prediction\"] = y_test\n# test[\"prediction\"].to_csv(f\"submission.csv\", index=True)","metadata":{},"execution_count":null,"outputs":[]}]}