{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:24:49.148655Z","iopub.execute_input":"2022-09-09T02:24:49.151366Z","iopub.status.idle":"2022-09-09T02:24:50.961365Z","shell.execute_reply.started":"2022-09-09T02:24:49.151223Z","shell.execute_reply":"2022-09-09T02:24:50.960332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Aggregated data are shared in dataset: https://www.kaggle.com/datasets/huseyincot/amex-agg-data-pickle\nData created with following code: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created","metadata":{}},{"cell_type":"code","source":"train = pd.read_pickle(\"../input/amex-agg-data-pickle/train_agg.pkl\", compression=\"gzip\")\ntest = pd.read_pickle(\"../input/amex-agg-data-pickle/test_agg.pkl\", compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:24:50.967785Z","iopub.execute_input":"2022-09-09T02:24:50.970602Z","iopub.status.idle":"2022-09-09T02:25:59.598522Z","shell.execute_reply.started":"2022-09-09T02:24:50.970543Z","shell.execute_reply":"2022-09-09T02:25:59.597565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:25:59.599705Z","iopub.execute_input":"2022-09-09T02:25:59.600077Z","iopub.status.idle":"2022-09-09T02:25:59.927316Z","shell.execute_reply.started":"2022-09-09T02:25:59.600048Z","shell.execute_reply":"2022-09-09T02:25:59.926233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competition Metric","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:25:59.929576Z","iopub.execute_input":"2022-09-09T02:25:59.929966Z","iopub.status.idle":"2022-09-09T02:25:59.944139Z","shell.execute_reply.started":"2022-09-09T02:25:59.929933Z","shell.execute_reply":"2022-09-09T02:25:59.942891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = test.columns.to_list()\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\"\n]\ncat_features = [f\"{cf}_last\" for cf in cat_features]\nle_encoder = LabelEncoder()\nfor categorical_feature in cat_features:\n    train[categorical_feature] = le_encoder.fit_transform(train[categorical_feature])\n    test[categorical_feature] = le_encoder.transform(test[categorical_feature])","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:25:59.945408Z","iopub.execute_input":"2022-09-09T02:25:59.946113Z","iopub.status.idle":"2022-09-09T02:26:02.290390Z","shell.execute_reply.started":"2022-09-09T02:25:59.946071Z","shell.execute_reply":"2022-09-09T02:26:02.289420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = pd.DataFrame(train[\"target\"])\ntrain_x = train.drop(\"target\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:26:02.291689Z","iopub.execute_input":"2022-09-09T02:26:02.292257Z","iopub.status.idle":"2022-09-09T02:26:07.693563Z","shell.execute_reply.started":"2022-09-09T02:26:02.292223Z","shell.execute_reply":"2022-09-09T02:26:07.692675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"N_FOLDS = 3\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(train_x.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(train_x, train_y):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        train_x.iloc[train_ind].reset_index(drop=True),\n        train_x.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        train_y.iloc[train_ind].reset_index(drop=True),\n        train_y.iloc[val_ind].reset_index(drop=True),\n    )\n\n    clf = CatBoostClassifier(iterations=5000, random_state=22)\n    clf.fit(tr_x, tr_y, eval_set=[(val_x, val_y)], cat_features=cat_features,  verbose=100)\n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = y_oof[val_ind] + preds\n\n    preds_test = clf.predict_proba(test)[:, 1]\n    y_test = y_test + preds_test / N_FOLDS\n    ix = ix + 1\ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y, y_pred)\nprint(f\"Amex metric: {val_score}\")","metadata":{"execution":{"iopub.status.busy":"2022-09-09T02:26:07.694618Z","iopub.execute_input":"2022-09-09T02:26:07.694947Z","iopub.status.idle":"2022-09-09T08:55:59.189048Z","shell.execute_reply.started":"2022-09-09T02:26:07.694917Z","shell.execute_reply":"2022-09-09T08:55:59.184516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_oof_binary = (y_oof >= np.percentile(y_oof, 96)).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T08:55:59.195984Z","iopub.execute_input":"2022-09-09T08:55:59.196664Z","iopub.status.idle":"2022-09-09T08:55:59.220887Z","shell.execute_reply.started":"2022-09-09T08:55:59.196598Z","shell.execute_reply":"2022-09-09T08:55:59.219905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_oof_binary.mean()","metadata":{"execution":{"iopub.status.busy":"2022-09-09T08:55:59.222618Z","iopub.execute_input":"2022-09-09T08:55:59.223207Z","iopub.status.idle":"2022-09-09T08:55:59.237050Z","shell.execute_reply.started":"2022-09-09T08:55:59.223155Z","shell.execute_reply":"2022-09-09T08:55:59.235899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot Confusion Matrix","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport itertools\nfrom sklearn.metrics import confusion_matrix\n\ndef plot_confusion_matrix(cm, classes,\n                          normalize = False,\n                          title = 'Confusion matrix\"',\n                          cmap = plt.cm.Blues) :\n    plt.imshow(cm, interpolation = 'nearest', cmap = cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation = 0)\n    plt.yticks(tick_marks, classes)\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])) :\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment = 'center',\n                 color = 'white' if cm[i, j] > thresh else 'black')\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    \ncm = confusion_matrix(train_y, y_oof_binary)\nclass_names = [0,1]\nplt.figure()\nplot_confusion_matrix(cm,\n                      classes = class_names,\n                      title = f'Confusion matrix at 4%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-09T08:55:59.241702Z","iopub.execute_input":"2022-09-09T08:55:59.242331Z","iopub.status.idle":"2022-09-09T08:55:59.763540Z","shell.execute_reply.started":"2022-09-09T08:55:59.242292Z","shell.execute_reply":"2022-09-09T08:55:59.762732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"test[\"prediction\"] = y_test\ntest[\"prediction\"].to_csv(f\"submission_cat_{val_score}.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T08:55:59.765066Z","iopub.execute_input":"2022-09-09T08:55:59.766073Z","iopub.status.idle":"2022-09-09T08:56:03.576031Z","shell.execute_reply.started":"2022-09-09T08:55:59.766033Z","shell.execute_reply":"2022-09-09T08:56:03.574769Z"},"trusted":true},"execution_count":null,"outputs":[]}]}