{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### I only made minimal changes (4 lines of code added) to the original [AMEX catboost notebook](https://www.kaggle.com/code/huseyincot/amex-catboost-0-793) Please upvote the original one made by https://www.kaggle.com/huseyincot ","metadata":{}},{"cell_type":"markdown","source":"### The idea is simple. Kaggle Grandmasters [RADDAR](https://www.kaggle.com/code/raddar/the-data-has-random-uniform-noise-added/notebook) and [Chris](https://www.kaggle.com/competitions/amex-default-prediction/discussion/327651) have observed that random noise has been added to data, whose magnitude is about `[0,0.01]`. Therefore we can simply round the data to the 2nd decimals to \"reduce\" this noise. It is helpful for tree models since they don't need to search for better splits within that noisy range `[0,0.01]` and thus reduces overfitting.","metadata":{}},{"cell_type":"markdown","source":"### CV score is improved to `0.7923` from `0.7905` and LB score is improved to `0.794` from `0.793`","metadata":{}},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-05-30T17:36:50.143861Z","iopub.execute_input":"2022-05-30T17:36:50.144392Z","iopub.status.idle":"2022-05-30T17:36:52.473529Z","shell.execute_reply.started":"2022-05-30T17:36:50.144272Z","shell.execute_reply":"2022-05-30T17:36:52.472667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Aggregated data are shared in dataset: https://www.kaggle.com/datasets/huseyincot/amex-agg-data-pickle\nData created with following code: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created","metadata":{}},{"cell_type":"code","source":"train = pd.read_pickle(\"../input/amex-agg-data-pickle/train_agg.pkl\", compression=\"gzip\")\ntest = pd.read_pickle(\"../input/amex-agg-data-pickle/test_agg.pkl\", compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-05-30T17:36:52.475215Z","iopub.execute_input":"2022-05-30T17:36:52.476552Z","iopub.status.idle":"2022-05-30T17:37:57.981546Z","shell.execute_reply.started":"2022-05-30T17:36:52.476499Z","shell.execute_reply":"2022-05-30T17:37:57.980225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The only change I made is the cell below.","metadata":{}},{"cell_type":"code","source":"for col in test.columns:\n    if test[col].dtype=='float16':\n        train[col] = train[col].astype('float32').round(decimals=2).astype('float16')\n        test[col] = test[col].astype('float32').round(decimals=2).astype('float16')","metadata":{"execution":{"iopub.status.busy":"2022-05-30T17:47:12.136025Z","iopub.execute_input":"2022-05-30T17:47:12.137187Z","iopub.status.idle":"2022-05-30T17:47:16.268766Z","shell.execute_reply.started":"2022-05-30T17:47:12.13714Z","shell.execute_reply":"2022-05-30T17:47:16.26799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competition Metric","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:03:56.130801Z","iopub.execute_input":"2022-05-28T11:03:56.131701Z","iopub.status.idle":"2022-05-28T11:03:56.145874Z","shell.execute_reply.started":"2022-05-28T11:03:56.131659Z","shell.execute_reply":"2022-05-28T11:03:56.144675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = test.columns.to_list()\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\"\n]\ncat_features = [f\"{cf}_last\" for cf in cat_features]\nle_encoder = LabelEncoder()\nfor categorical_feature in cat_features:\n    train[categorical_feature] = le_encoder.fit_transform(train[categorical_feature])\n    test[categorical_feature] = le_encoder.transform(test[categorical_feature])","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:04:24.704313Z","iopub.execute_input":"2022-05-28T11:04:24.704836Z","iopub.status.idle":"2022-05-28T11:04:27.118882Z","shell.execute_reply.started":"2022-05-28T11:04:24.704799Z","shell.execute_reply":"2022-05-28T11:04:27.118011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = pd.DataFrame(train[\"target\"])\ntrain_x = train.drop(\"target\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:04:31.584929Z","iopub.execute_input":"2022-05-28T11:04:31.585549Z","iopub.status.idle":"2022-05-28T11:04:34.311936Z","shell.execute_reply.started":"2022-05-28T11:04:31.585511Z","shell.execute_reply":"2022-05-28T11:04:34.310872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"N_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(train_x.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(train_x, train_y):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        train_x.iloc[train_ind].reset_index(drop=True),\n        train_x.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        train_y.iloc[train_ind].reset_index(drop=True),\n        train_y.iloc[val_ind].reset_index(drop=True),\n    )\n\n    clf = CatBoostClassifier(iterations=5000, random_state=22)\n    clf.fit(tr_x, tr_y, eval_set=[(val_x, val_y)], cat_features=cat_features,  verbose=100)\n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = y_oof[val_ind] + preds\n\n    preds_test = clf.predict_proba(test)[:, 1]\n    y_test = y_test + preds_test / N_FOLDS\n    ix = ix + 1\ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y, y_pred)\nprint(f\"Amex metric: {val_score}\")","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:04:37.189655Z","iopub.execute_input":"2022-05-28T11:04:37.190057Z","iopub.status.idle":"2022-05-28T11:48:09.281003Z","shell.execute_reply.started":"2022-05-28T11:04:37.190028Z","shell.execute_reply":"2022-05-28T11:48:09.279487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_oof_binary = (y_oof >= np.percentile(y_oof, 96)).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:48:40.202001Z","iopub.execute_input":"2022-05-28T11:48:40.202711Z","iopub.status.idle":"2022-05-28T11:48:40.215705Z","shell.execute_reply.started":"2022-05-28T11:48:40.202669Z","shell.execute_reply":"2022-05-28T11:48:40.214492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_oof_binary.mean()","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:48:42.035844Z","iopub.execute_input":"2022-05-28T11:48:42.036509Z","iopub.status.idle":"2022-05-28T11:48:42.044473Z","shell.execute_reply.started":"2022-05-28T11:48:42.03645Z","shell.execute_reply":"2022-05-28T11:48:42.043297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot Confusion Matrix","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport itertools\nfrom sklearn.metrics import confusion_matrix\n\ndef plot_confusion_matrix(cm, classes,\n                          normalize = False,\n                          title = 'Confusion matrix\"',\n                          cmap = plt.cm.Blues) :\n    plt.imshow(cm, interpolation = 'nearest', cmap = cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation = 0)\n    plt.yticks(tick_marks, classes)\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])) :\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment = 'center',\n                 color = 'white' if cm[i, j] > thresh else 'black')\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    \ncm = confusion_matrix(train_y, y_oof_binary)\nclass_names = [0,1]\nplt.figure()\nplot_confusion_matrix(cm,\n                      classes = class_names,\n                      title = f'Confusion matrix at 4%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-28T11:49:17.466134Z","iopub.execute_input":"2022-05-28T11:49:17.467989Z","iopub.status.idle":"2022-05-28T11:49:17.883413Z","shell.execute_reply.started":"2022-05-28T11:49:17.467931Z","shell.execute_reply":"2022-05-28T11:49:17.882427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"test[\"prediction\"] = y_test\ntest[\"prediction\"].to_csv(f\"submission_cat_{val_score}.csv\", index=True)","metadata":{},"execution_count":null,"outputs":[]}]}