{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":3709574,"sourceType":"datasetVersion","datasetId":2217540}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:15:21.644793Z","iopub.execute_input":"2025-08-23T05:15:21.645076Z","iopub.status.idle":"2025-08-23T05:15:23.589492Z","shell.execute_reply.started":"2025-08-23T05:15:21.645054Z","shell.execute_reply":"2025-08-23T05:15:23.588915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_pickle(\"../input/amex-agg-data-pickle/train_agg.pkl\", compression=\"gzip\")\ntest = pd.read_pickle(\"../input/amex-agg-data-pickle/test_agg.pkl\", compression=\"gzip\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:15:25.864594Z","iopub.execute_input":"2025-08-23T05:15:25.864951Z","iopub.status.idle":"2025-08-23T05:16:15.360936Z","shell.execute_reply.started":"2025-08-23T05:15:25.864931Z","shell.execute_reply":"2025-08-23T05:16:15.360339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:16:26.491467Z","iopub.execute_input":"2025-08-23T05:16:26.492076Z","iopub.status.idle":"2025-08-23T05:16:26.514176Z","shell.execute_reply.started":"2025-08-23T05:16:26.492043Z","shell.execute_reply":"2025-08-23T05:16:26.513362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:16:41.274613Z","iopub.execute_input":"2025-08-23T05:16:41.274915Z","iopub.status.idle":"2025-08-23T05:16:41.283908Z","shell.execute_reply.started":"2025-08-23T05:16:41.274893Z","shell.execute_reply":"2025-08-23T05:16:41.283128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = test.columns.to_list()\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\"\n]\ncat_features = [f\"{cf}_last\" for cf in cat_features]\nle_encoder = LabelEncoder()\nfor categorical_feature in cat_features:\n    train[categorical_feature] = le_encoder.fit_transform(train[categorical_feature])\n    test[categorical_feature] = le_encoder.transform(test[categorical_feature])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:16:45.324430Z","iopub.execute_input":"2025-08-23T05:16:45.325259Z","iopub.status.idle":"2025-08-23T05:16:46.664546Z","shell.execute_reply.started":"2025-08-23T05:16:45.325228Z","shell.execute_reply":"2025-08-23T05:16:46.664000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_y = pd.DataFrame(train[\"target\"])\ntrain_x = train.drop(\"target\", axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:16:50.654457Z","iopub.execute_input":"2025-08-23T05:16:50.655032Z","iopub.status.idle":"2025-08-23T05:16:52.677868Z","shell.execute_reply.started":"2025-08-23T05:16:50.655007Z","shell.execute_reply":"2025-08-23T05:16:52.677283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(train_x.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(train_x, train_y):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        train_x.iloc[train_ind].reset_index(drop=True),\n        train_x.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        train_y.iloc[train_ind].reset_index(drop=True),\n        train_y.iloc[val_ind].reset_index(drop=True),\n    )\n\n    clf = CatBoostClassifier(iterations=5000, random_state=22)\n    clf.fit(tr_x, tr_y, eval_set=[(val_x, val_y)], cat_features=cat_features,  verbose=100)\n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = y_oof[val_ind] + preds\n\n    preds_test = clf.predict_proba(test)[:, 1]\n    y_test = y_test + preds_test / N_FOLDS\n    ix = ix + 1\ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y, y_pred)\nprint(f\"Amex metric: {val_score}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T05:16:56.055256Z","iopub.execute_input":"2025-08-23T05:16:56.055638Z","iopub.status.idle":"2025-08-23T12:08:45.106428Z","shell.execute_reply.started":"2025-08-23T05:16:56.055609Z","shell.execute_reply":"2025-08-23T12:08:45.105579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_oof_binary = (y_oof >= np.percentile(y_oof, 96)).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T12:13:47.097726Z","iopub.execute_input":"2025-08-23T12:13:47.098437Z","iopub.status.idle":"2025-08-23T12:13:47.110595Z","shell.execute_reply.started":"2025-08-23T12:13:47.098413Z","shell.execute_reply":"2025-08-23T12:13:47.109999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_oof_binary.mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T12:13:53.107746Z","iopub.execute_input":"2025-08-23T12:13:53.108026Z","iopub.status.idle":"2025-08-23T12:13:53.113791Z","shell.execute_reply.started":"2025-08-23T12:13:53.108004Z","shell.execute_reply":"2025-08-23T12:13:53.113123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport itertools\nfrom sklearn.metrics import confusion_matrix\n\ndef plot_confusion_matrix(cm, classes,\n                          normalize = False,\n                          title = 'Confusion matrix\"',\n                          cmap = plt.cm.Blues) :\n    plt.imshow(cm, interpolation = 'nearest', cmap = cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation = 0)\n    plt.yticks(tick_marks, classes)\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])) :\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment = 'center',\n                 color = 'white' if cm[i, j] > thresh else 'black')\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    \ncm = confusion_matrix(train_y, y_oof_binary)\nclass_names = [0,1]\nplt.figure()\nplot_confusion_matrix(cm,\n                      classes = class_names,\n                      title = f'Confusion matrix at 4%')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T12:13:59.438144Z","iopub.execute_input":"2025-08-23T12:13:59.438448Z","iopub.status.idle":"2025-08-23T12:14:00.004499Z","shell.execute_reply.started":"2025-08-23T12:13:59.438428Z","shell.execute_reply":"2025-08-23T12:14:00.003783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\"prediction\": y_test}, index=test.index)\nsubmission.to_csv(f\"submission_cat_{val_score}.csv\", index=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-23T12:16:07.437608Z","iopub.execute_input":"2025-08-23T12:16:07.437881Z","iopub.status.idle":"2025-08-23T12:16:10.952052Z","shell.execute_reply.started":"2025-08-23T12:16:07.437862Z","shell.execute_reply":"2025-08-23T12:16:10.951245Z"}},"outputs":[],"execution_count":null}]}