{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Kemal Azamat (HSE, CDM) \n\n[Kaggle](https://www.kaggle.com/c/home-credit-default-risk) \n\nIn updating this notebook, I made focused changes to enhance the model's performance and future readiness:\n\n1. Improved Categorical Handling: Implemented efficient encoding and handling of unseen categories in test data.\n\n2. Advanced Feature Engineering: Introduced sophisticated aggregation features for a deeper understanding of customer behavior.\n\n3. Model Transition: Moved to XGBoost for its superior performance, with fine-tuned hyperparameters for optimal balance.\n\n4. Temporal Stability Evaluation: Added stability scoring to assess the model's consistency over time, ensuring future reliability.\n\n5. Streamlined Submission: Refined the submission preparation process, ensuring compatibility with competition guidelines and test data nuances.\n\n## Load the data","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.manifold import TSNE\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:04.161178Z","iopub.execute_input":"2024-03-19T20:24:04.161808Z","iopub.status.idle":"2024-03-19T20:24:07.905996Z","shell.execute_reply.started":"2024-03-19T20:24:04.161754Z","shell.execute_reply":"2024-03-19T20:24:07.904455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:07.909005Z","iopub.execute_input":"2024-03-19T20:24:07.909491Z","iopub.status.idle":"2024-03-19T20:24:07.921071Z","shell.execute_reply.started":"2024-03-19T20:24:07.909450Z","shell.execute_reply":"2024-03-19T20:24:07.919868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:07.923054Z","iopub.execute_input":"2024-03-19T20:24:07.923819Z","iopub.status.idle":"2024-03-19T20:24:27.697372Z","shell.execute_reply.started":"2024-03-19T20:24:07.923780Z","shell.execute_reply":"2024-03-19T20:24:27.696145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:27.701542Z","iopub.execute_input":"2024-03-19T20:24:27.702090Z","iopub.status.idle":"2024-03-19T20:24:27.779310Z","shell.execute_reply.started":"2024-03-19T20:24:27.702038Z","shell.execute_reply":"2024-03-19T20:24:27.777834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering\n\nIn this part, we can see a simple example of joining tables via `case_id`. Here the loading and joining is done with polars library. Polars library is blazingly fast and has much smaller memory footprint than pandas. ","metadata":{}},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:27.784677Z","iopub.execute_input":"2024-03-19T20:24:27.785408Z","iopub.status.idle":"2024-03-19T20:24:29.941025Z","shell.execute_reply.started":"2024-03-19T20:24:27.785348Z","shell.execute_reply":"2024-03-19T20:24:29.939884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:29.942203Z","iopub.execute_input":"2024-03-19T20:24:29.942684Z","iopub.status.idle":"2024-03-19T20:24:29.958930Z","shell.execute_reply.started":"2024-03-19T20:24:29.942639Z","shell.execute_reply":"2024-03-19T20:24:29.957657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"case_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\nprint(cols_pred)\n\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:29.960750Z","iopub.execute_input":"2024-03-19T20:24:29.961270Z","iopub.status.idle":"2024-03-19T20:24:39.520260Z","shell.execute_reply.started":"2024-03-19T20:24:29.961235Z","shell.execute_reply":"2024-03-19T20:24:39.518709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:39.521943Z","iopub.execute_input":"2024-03-19T20:24:39.522352Z","iopub.status.idle":"2024-03-19T20:24:39.530369Z","shell.execute_reply.started":"2024-03-19T20:24:39.522317Z","shell.execute_reply":"2024-03-19T20:24:39.528970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:39.532382Z","iopub.execute_input":"2024-03-19T20:24:39.533121Z","iopub.status.idle":"2024-03-19T20:24:39.546433Z","shell.execute_reply.started":"2024-03-19T20:24:39.533075Z","shell.execute_reply":"2024-03-19T20:24:39.544898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PCA Visualization Steps\n\n1. **Select Numerical Features**: Extract only the numerical columns from the training data.\n\n2. **Select Categorical Features**: Identify and extract categorical columns from the training data.\n\n3. **One-Hot Encoding**: Apply One-Hot Encoding to the categorical features to convert them into a numerical format, handling any unknown categories encountered in future data.\n\n4. **Imputation**: Impute missing values in numerical features using the mean of each column to ensure the model receives a complete dataset.\n\n5. **Feature Combination**: Combine the imputed numerical features with the one-hot encoded categorical features to prepare the final dataset for PCA.\n\n6. **PCA Application**: Apply Principal Component Analysis (PCA) to reduce the dimensionality of the dataset, focusing on capturing the most significant variance with a limited number of components.\n\n7. **PCA Visualization**: Visualize the transformed dataset in a 2D space, using the first two principal components, and color the points based on their corresponding target values to observe potential clusters or patterns.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\n\nX_train_numeric = X_train.select_dtypes(include=[np.number])\nX_train_categorical = X_train.select_dtypes(include=['object', 'category'])\n\n\nif not X_train_categorical.empty:\n\n    encoder = OneHotEncoder(sparse=False, handle_unknown='ignore')  \n    \n    X_train_encoded = encoder.fit_transform(X_train_categorical)\n\n    num_imputer = SimpleImputer(strategy='mean')\n    X_train_numeric_imputed = num_imputer.fit_transform(X_train_numeric)\n\n\n    X_train_prepared = np.concatenate((X_train_numeric_imputed, X_train_encoded), axis=1)\nelse:\n\n\n    num_imputer = SimpleImputer(strategy='mean')\n    X_train_prepared = num_imputer.fit_transform(X_train_numeric)\n\n\npca = PCA(n_components=2, random_state=0)\n\nX_pca = pca.fit_transform(X_train_prepared)\n\n\nplt.figure(figsize=(10, 5))\nplt.scatter(X_pca[:, 0], X_pca[:, 1], c=y_train)\nplt.title('PCA visualization')\nplt.xlabel('First main')\nplt.ylabel('Second main')\nplt.colorbar()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:24:39.548368Z","iopub.execute_input":"2024-03-19T20:24:39.548832Z","iopub.status.idle":"2024-03-19T20:26:23.464458Z","shell.execute_reply.started":"2024-03-19T20:24:39.548792Z","shell.execute_reply":"2024-03-19T20:26:23.463229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing missing values","metadata":{}},{"cell_type":"code","source":"import missingno as msno\n\nmsno.matrix(X_train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:26:23.466463Z","iopub.execute_input":"2024-03-19T20:26:23.467267Z","iopub.status.idle":"2024-03-19T20:26:34.305554Z","shell.execute_reply.started":"2024-03-19T20:26:23.467219Z","shell.execute_reply":"2024-03-19T20:26:34.303776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### For a more detailed analysis","metadata":{}},{"cell_type":"code","source":"msno.bar(X_train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:26:34.307954Z","iopub.execute_input":"2024-03-19T20:26:34.308781Z","iopub.status.idle":"2024-03-19T20:26:38.064031Z","shell.execute_reply.started":"2024-03-19T20:26:34.308730Z","shell.execute_reply":"2024-03-19T20:26:38.061880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of numerical signs","metadata":{}},{"cell_type":"code","source":"sns.histplot(X_train['avgpmtlast12m_4525200A'], kde=True, bins=30)\nplt.title('Distribution of average payments for the last 12 months')\nplt.show()\n\n\nsns.boxplot(data=X_train[['credamount_770A', 'currdebt_22A']])\nplt.title('Boxplot of credit amounts and current debts')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:26:38.065887Z","iopub.execute_input":"2024-03-19T20:26:38.067022Z","iopub.status.idle":"2024-03-19T20:26:41.776821Z","shell.execute_reply.started":"2024-03-19T20:26:38.066925Z","shell.execute_reply":"2024-03-19T20:26:41.775160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of categorical attributes","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='education_88M', data=X_train)\nplt.title('Distribution of educational categories')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:26:41.783498Z","iopub.execute_input":"2024-03-19T20:26:41.784122Z","iopub.status.idle":"2024-03-19T20:26:42.037674Z","shell.execute_reply.started":"2024-03-19T20:26:41.784067Z","shell.execute_reply":"2024-03-19T20:26:42.036096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = X_train.select_dtypes(include=[np.number]).corr()\nsns.heatmap(corr, annot=False, cmap='coolwarm')\nplt.title('Correlation matrix of numerical features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:26:42.039269Z","iopub.execute_input":"2024-03-19T20:26:42.039634Z","iopub.status.idle":"2024-03-19T20:26:45.285333Z","shell.execute_reply.started":"2024-03-19T20:26:42.039602Z","shell.execute_reply":"2024-03-19T20:26:45.283943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost + LabelEncoder + hyperopt\n\nTraining","metadata":{}},{"cell_type":"code","source":"# from sklearn.preprocessing import LabelEncoder\n\n\n# label_encoder = LabelEncoder()\n# label_encoder.fit(list(X_train[col].unique()) + ['unknown'])\n\n# for col in X_train.select_dtypes(include=['category', 'object']).columns:\n#     X_train[col] = X_train[col].cat.add_categories(['missing'])\n#     X_valid[col] = X_valid[col].cat.add_categories(['missing'])\n\n#     #  NaN -> 'missing'\n#     X_train[col] = X_train[col].fillna('missing').astype(str)\n#     X_valid[col] = X_valid[col].fillna('missing').astype(str)\n\n#     X_train[col] = X_train[col].fillna('missing').astype(str)\n#     X_valid[col] = X_valid[col].fillna('missing').astype(str)\n    \n\n#     combined_categories = np.union1d(X_train[col].unique(), X_valid[col].unique())\n    \n\n#     combined_categories_with_unknown = np.append(combined_categories, 'unknown')\n    \n#     label_encoder.fit(combined_categories_with_unknown)\n    \n#     X_train[col] = label_encoder.transform(X_train[col])\n#     X_valid[col] = label_encoder.transform(X_valid[col])\n","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:39:56.400254Z","iopub.execute_input":"2024-03-19T20:39:56.400774Z","iopub.status.idle":"2024-03-19T20:40:01.464689Z","shell.execute_reply.started":"2024-03-19T20:39:56.400714Z","shell.execute_reply":"2024-03-19T20:40:01.463334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train_processed = X_train.copy()\n# X_valid_processed = X_valid.copy()\n\n# label_encoder = LabelEncoder()\n\n\n# for col in X_train.select_dtypes(include=['object', 'category']).columns:\n\n#     combined_data = pd.concat([X_train[col], X_valid[col]], axis=0)\n#     label_encoder.fit(combined_data)\n    \n\n#     X_train_processed[col] = label_encoder.transform(X_train[col])\n#     X_valid_processed[col] = label_encoder.transform(X_valid[col])","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:45:26.278009Z","iopub.execute_input":"2024-03-19T20:45:26.278505Z","iopub.status.idle":"2024-03-19T20:45:26.285452Z","shell.execute_reply.started":"2024-03-19T20:45:26.278470Z","shell.execute_reply":"2024-03-19T20:45:26.283682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from hyperopt import hp, fmin, tpe, STATUS_OK, Trials\n# from sklearn.metrics import roc_auc_score\n\n\n# space = {\n#     'max_depth': hp.choice('max_depth', range(1, 3, 1)),\n#     'min_child_weight': hp.quniform('min_child_weight', 1, 5, 1),\n#     'subsample': hp.uniform('subsample', 0.5, 1),\n#     'colsample_bytree': hp.uniform('colsample_bytree', 0.5, 1),\n#     'learning_rate': hp.uniform('learning_rate', 0.01, 0.2),\n#     'n_estimators': hp.choice('n_estimators', range(100, 1100, 100)),\n#     'objective': 'binary:logistic',\n#     'eval_metric': 'auc',\n#     'booster': 'gbtree',\n#     'seed': 0\n# }\n\n\n# def objective(space):\n#     clf = xgb.XGBClassifier(\n#         n_estimators = int(space['n_estimators']), max_depth = int(space['max_depth']),\n#         min_child_weight = space['min_child_weight'], subsample = space['subsample'],\n#         colsample_bytree = space['colsample_bytree'], learning_rate = space['learning_rate'],\n#         objective = space['objective'], booster = space['booster'], eval_metric = space['eval_metric'],\n#         use_label_encoder=False\n#     )\n    \n#     evaluation = [(X_train_processed, y_train), (X_valid_processed, y_valid)]\n    \n#     clf.fit(X_train_processed, y_train, eval_set=evaluation, early_stopping_rounds=10, verbose=False)\n\n#     pred = clf.predict_proba(X_valid)[:,1]\n#     auc = roc_auc_score(y_valid, pred)\n#     print(f\"AUC: {auc}\")\n#     return {'loss': -auc, 'status': STATUS_OK }\n\n# trials = Trials()\n# best_hyperparams = fmin(fn=objective, space=space, algo=tpe.suggest, max_evals=100, trials=trials)\n\n# print(\"Best:\", best_hyperparams)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:45:30.339877Z","iopub.execute_input":"2024-03-19T20:45:30.340444Z","iopub.status.idle":"2024-03-19T20:45:30.348926Z","shell.execute_reply.started":"2024-03-19T20:45:30.340400Z","shell.execute_reply":"2024-03-19T20:45:30.347473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.metrics import roc_auc_score\n\n\n\ndtrain = xgb.DMatrix(X_train, label=y_train, enable_categorical=True)\ndvalid = xgb.DMatrix(X_valid, label=y_valid, enable_categorical=True)\ndtest = xgb.DMatrix(X_test, enable_categorical=True)\n\n\nxgb_params = {\n    'booster': 'gbtree',\n    'objective': 'binary:logistic',\n    'eval_metric': 'auc',\n    'learning_rate': 0.05,\n    'max_depth': 2,\n    'min_child_weight': 1,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'n_estimators': 1000,\n    \"bagging_freq\": 5,\n    'seed': 0,\n    'enable_categorical': True,\n}\n\n\nevals_result = {}\n\nxgb_model = xgb.train(\n    params=xgb_params,\n    dtrain=dtrain,\n    num_boost_round=1000,\n    early_stopping_rounds=10,\n    evals=[(dvalid, 'valid')],\n    verbose_eval=50,\n    evals_result=evals_result\n)\n\n\nepochs = len(evals_result['valid']['auc'])\nx_axis = range(0, epochs)\n\nplt.plot(x_axis, evals_result['valid']['auc'], label='Valid')\nplt.legend()\nplt.ylabel('AUC')\nplt.xlabel('Round of Boosting')\nplt.title('XGBoost AUC')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:29:49.690353Z","iopub.execute_input":"2024-03-19T20:29:49.690826Z","iopub.status.idle":"2024-03-19T20:29:56.113263Z","shell.execute_reply.started":"2024-03-19T20:29:49.690793Z","shell.execute_reply":"2024-03-19T20:29:56.111310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_train = xgb_model.predict(dtrain)\ny_pred_valid = xgb_model.predict(dvalid)\ny_pred_test = xgb_model.predict(dtest)\n\n\nprint(f'The AUC score on the train set is: {roc_auc_score(y_train, y_pred_train)}')\nprint(f'The AUC score on the valid set is: {roc_auc_score(y_valid, y_pred_valid)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:04.031471Z","iopub.execute_input":"2024-03-19T20:30:04.031965Z","iopub.status.idle":"2024-03-19T20:30:04.731864Z","shell.execute_reply.started":"2024-03-19T20:30:04.031929Z","shell.execute_reply":"2024-03-19T20:30:04.730854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = xgb.plot_importance(xgb_model, max_num_features=10, height=0.8)\nax.grid(False)\nax.set_title('Top 10 Important Features with XGBoost')\nax.set_xlabel('F-Score')\nax.set_ylabel('Features')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:05.800889Z","iopub.execute_input":"2024-03-19T20:30:05.801364Z","iopub.status.idle":"2024-03-19T20:30:06.097362Z","shell.execute_reply.started":"2024-03-19T20:30:05.801329Z","shell.execute_reply":"2024-03-19T20:30:06.096144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluation with AUC and then comparison with the stability metric is shown below.","metadata":{}},{"cell_type":"code","source":"for base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    dmatrix = xgb.DMatrix(X, enable_categorical=True)\n    y_pred = xgb_model.predict(dmatrix)\n    base[\"score\"] = y_pred\n\nauc_train = roc_auc_score(base_train[\"target\"], base_train[\"score\"])\nauc_valid = roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])\nauc_test = roc_auc_score(base_test[\"target\"], base_test[\"score\"])\n\nprint(f'The AUC score on the train set is: {auc_train:.2f}')\nprint(f'The AUC score on the valid set is: {auc_valid:.2f}')\nprint(f'The AUC score on the test set is: {auc_test:.2f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:09.040020Z","iopub.execute_input":"2024-03-19T20:30:09.040464Z","iopub.status.idle":"2024-03-19T20:30:11.755306Z","shell.execute_reply.started":"2024-03-19T20:30:09.040428Z","shell.execute_reply":"2024-03-19T20:30:11.754004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}')  ","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:16.379706Z","iopub.execute_input":"2024-03-19T20:30:16.380213Z","iopub.status.idle":"2024-03-19T20:30:17.479662Z","shell.execute_reply.started":"2024-03-19T20:30:16.380175Z","shell.execute_reply":"2024-03-19T20:30:17.478047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\n\nScoring the submission dataset is below, we need to take care of new categories. Then we save the score as a last step. ","metadata":{}},{"cell_type":"code","source":"X_submission = data_submission[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)\n\n\ncategorical_cols = X_train.select_dtypes(include=['category']).columns\n\n\nfor col in categorical_cols:\n    if X_submission[col].dtype.name == 'category':\n        X_submission[col] = X_submission[col].cat.codes \n\n\ncategorical_cols_in_submission = X_submission.select_dtypes(include=['category']).columns\nif len(categorical_cols_in_submission) > 0:\n    print(f\"Attention: in X_submission are some categorials: {list(categorical_cols_in_submission)}\")\n\n\n\ndsubmission = xgb.DMatrix(X_submission, enable_categorical=True)\n\n\ny_submission_pred = xgb_model.predict(dsubmission)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:20.309634Z","iopub.execute_input":"2024-03-19T20:30:20.310138Z","iopub.status.idle":"2024-03-19T20:30:20.381139Z","shell.execute_reply.started":"2024-03-19T20:30:20.310101Z","shell.execute_reply":"2024-03-19T20:30:20.380070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submission_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:22.579833Z","iopub.execute_input":"2024-03-19T20:30:22.580271Z","iopub.status.idle":"2024-03-19T20:30:22.589805Z","shell.execute_reply.started":"2024-03-19T20:30:22.580234Z","shell.execute_reply":"2024-03-19T20:30:22.588474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-19T20:30:23.680408Z","iopub.execute_input":"2024-03-19T20:30:23.681137Z","iopub.status.idle":"2024-03-19T20:30:23.689822Z","shell.execute_reply.started":"2024-03-19T20:30:23.681100Z","shell.execute_reply":"2024-03-19T20:30:23.688778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}