{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Author: Wu Zhaofeng wuzhaofeng2001@163.com;\n\n#        Yang Junna","metadata":{}},{"cell_type":"markdown","source":"# Import Packages and define Data Conversion function","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import GridSearchCV\nimport lightgbm as lgb\nfrom sklearn import metrics\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, log_loss\nimport warnings\nimport pyarrow as pa\nwarnings.simplefilter(action='ignore', category=FutureWarning)\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:09.985580Z","iopub.execute_input":"2024-03-10T03:57:09.986670Z","iopub.status.idle":"2024-03-10T03:57:11.890083Z","shell.execute_reply.started":"2024-03-10T03:57:09.986626Z","shell.execute_reply":"2024-03-10T03:57:11.888767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame):\n    # implement here all desired dtypes for tables\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        #A is amount; P is percentage;\n        if col[-1] in (\"P\", \"A\"):\n            #Store data in floating data type and name it with original name\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n    return df\n\ndef convert_strings(df: pd.DataFrame):\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            #Convert Data into string data, and then convert them into categorical data.\n            df[col] = df[col].astype(\"string\").astype('category')\n            #Get current_categories ,namely the unique values of categories.\n            current_categories = df[col].cat.categories\n            #Null data will be 'Unknown'\n            #Unknown is used for new type not in the training set so we do not need to clear new type\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            #Convert list into categorical data with order\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            #Replace values\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:14.530455Z","iopub.execute_input":"2024-03-10T03:57:14.531925Z","iopub.status.idle":"2024-03-10T03:57:14.541744Z","shell.execute_reply.started":"2024-03-10T03:57:14.531878Z","shell.execute_reply":"2024-03-10T03:57:14.540713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Import Train Dataset**\n'Pipe' can apply the function on the table.","metadata":{}},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\n#Vertical join other training set table and convert numeric data into floating type\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:16.330333Z","iopub.execute_input":"2024-03-10T03:57:16.331678Z","iopub.status.idle":"2024-03-10T03:57:33.803940Z","shell.execute_reply.started":"2024-03-10T03:57:16.331625Z","shell.execute_reply":"2024-03-10T03:57:33.802783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Test Dataset","metadata":{}},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:36.058389Z","iopub.execute_input":"2024-03-10T03:57:36.058861Z","iopub.status.idle":"2024-03-10T03:57:36.105903Z","shell.execute_reply.started":"2024-03-10T03:57:36.058825Z","shell.execute_reply":"2024-03-10T03:57:36.104667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering\n(1)Group by case_id so the record of the same client will be together.\n\n(2)Use aggregation function:Create new columns with the max value respectively.\n\n(3)When num_group == 0, it is the applicant (the person who applied for a loan. So we only select target customers.\n\n(4)Join tables on case_id.\n","metadata":{}},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\n#mainoccupationinc_384A is main income； Choose Selefemployed persons' max income.\n#pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\") will return bool indicating whether it is selfemployed and return the number of True via max() in certain case_id\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\"), #29199 True\n    (pl.col(\"incometype_1044T\") == \"EMPLOYED\").max().alias(\"mainoccupationinc_384A_employed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\n#pmts_pmtsoverdue_635A:Active contract that has overdue payments (num_group1 - existing contract, num_group2 - payment).\n#pmts_dpdvalue_108P:Value of past due payment for active contract (num_group1 - existing contract, num_group2 - payment).\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# A type is amount; M type is string(mainly reasons);L type is category(like type of applicants, type).\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\",'L'):\n        selected_static_cols.append(col)\n#print(selected_static_cols)\n#print()\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\",'L'):\n        selected_static_cb_cols.append(col)\n#print(selected_static_cb_cols)\n\n# Join all tables on case_id.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:38.813891Z","iopub.execute_input":"2024-03-10T03:57:38.814307Z","iopub.status.idle":"2024-03-10T03:57:42.469644Z","shell.execute_reply.started":"2024-03-10T03:57:38.814273Z","shell.execute_reply":"2024-03-10T03:57:42.468330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We still create a subset only have M type and A type data.","metadata":{}},{"cell_type":"code","source":"subset_cols_1 = [col for col in data.columns if col.endswith(('M', 'A'))]\nsubset_cols_2=['case_id','date_decision','MONTH','target',\"person_housetype\",\"pmts_dpdvalue_108P_over31\"]\nsubset_cols=subset_cols_2 + subset_cols_1\nsubset_data_train = data[subset_cols]","metadata":{"execution":{"iopub.status.busy":"2024-03-10T00:39:19.220246Z","iopub.execute_input":"2024-03-10T00:39:19.220681Z","iopub.status.idle":"2024-03-10T00:39:19.227593Z","shell.execute_reply.started":"2024-03-10T00:39:19.220635Z","shell.execute_reply":"2024-03-10T00:39:19.226708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\"),\n    (pl.col(\"incometype_1044T\") == \"EMPLOYED\").max().alias(\"mainoccupationinc_384A_employed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n#Here we use the column name generated on training set\n#This test set is used for prediction\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:45.451815Z","iopub.execute_input":"2024-03-10T03:57:45.452226Z","iopub.status.idle":"2024-03-10T03:57:45.475018Z","shell.execute_reply.started":"2024-03-10T03:57:45.452196Z","shell.execute_reply":"2024-03-10T03:57:45.473977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Be careful. There is no 'target' variable in test set.\nsubset_cols_1 = [col for col in data_submission.columns if col.endswith(('M', 'A'))]\nsubset_cols_2=['case_id','date_decision','MONTH',\"person_housetype\",\"pmts_dpdvalue_108P_over31\"]\nsubset_cols=subset_cols_2 + subset_cols_1\nsubset_data_test = data_submission[subset_cols]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define training set and test set","metadata":{}},{"cell_type":"code","source":"#60% train set, 20% validation set, 20% test set.\ncase_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_list = case_ids.to_list()\ncase_ids_train, case_ids_test = train_test_split(case_ids_list, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\n#If the column names are ending with upper letther and the other characters are lower, then add them in the list. \ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\n#print(cols_pred)\n#Observations without case_id will be filtered. And we get 3 sets of variables: base_train used for Gini index,x,y.\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:57:51.805659Z","iopub.execute_input":"2024-03-10T03:57:51.806115Z","iopub.status.idle":"2024-03-10T03:58:24.325575Z","shell.execute_reply.started":"2024-03-10T03:57:51.806076Z","shell.execute_reply":"2024-03-10T03:58:24.324253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Here we see the n of sets and check if they have the same length.\nprint(f\"Base_Train: {base_train.shape}\")\nprint(f\"X_Train: {X_train.shape}\")\nprint(f\"y_Train: {y_train.shape}\")\nprint(f\"X_Valid: {X_valid.shape}\")\nprint(f\"X_Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-10T02:48:13.626296Z","iopub.execute_input":"2024-03-10T02:48:13.627566Z","iopub.status.idle":"2024-03-10T02:48:13.634812Z","shell.execute_reply.started":"2024-03-10T02:48:13.627510Z","shell.execute_reply":"2024-03-10T02:48:13.633384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LightGBM on training set[](http://)\npredict_y = sum of fk(xi) from k=1 to K;\n\n\nDelta Loss = Residual_1^2/n1+Residual_2^2/n2-Residual_ParentSet^2/n_Parent\n\nGOSS first sorts the data based on the absolute gradient values, selects the top a instances, and then randomly samples b instances from the remaining data. This approach allows the algorithm to focus more on under-trained instances without significantly altering the original dataset's distribution. GOSS retains all high-gradient samples, randomly samples low-gradient samples, and to ensure distribution consistency, when calculating information gain, the sampled low-gradient samples are multiplied by a constant factor: (1−a)/b. Here, a represents the proportion of the top a×100% high-gradient samples, b represents the sampling ratio of low-gradient samples, the multiplication by 1−a is because the overall sampling of high-gradient samples covers the entire sample set N, while the candidate sample set for low-gradient sampling is (1−a)*N, and division by b is necessary to compensate for the reduced overall distribution of the low-gradient samples due to sampling, thereby requiring a weight amplification of 1/b.\n\nEFB combine similar features into a new feature.\n\nHistogram-based Splitting：first discretizes continuous features into bins. This is done to reduce the number of unique feature values, lowering memory usage and computational complexity.\n\nLeaf-wise Tree Growth:Choose the maximum improvement tree to split.\n\nGradient Boosting:improve residual via iteration\n\nRegularization","metadata":{}},{"cell_type":"code","source":"lgb_train = lgb.Dataset(X_train, label=y_train)\nlgb_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_train)\n#Here max_depth and learning_rate is determined by cross validation\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"num_leaves\": 30,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}\n\ngbm = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,\n    callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:41:40.308651Z","iopub.execute_input":"2024-03-10T03:41:40.309060Z","iopub.status.idle":"2024-03-10T03:45:17.665793Z","shell.execute_reply.started":"2024-03-10T03:41:40.309030Z","shell.execute_reply":"2024-03-10T03:45:17.664709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross Validation\n### `max_depth`、`num_leaves` and`learning_rate` are the most important hyperparameters.The first two terms will control the complexity of the model. And the last one will control the rounds of iteration.\nAnyway,we only have limited GPU and time.The polar package is not compatible with the pyarrow on TPU/GPU offered by Kaggle. So we just consider learning rate here. But a great improvement will be obtained if we take 3 hyperparameters into consideration.\nDo not run this code unless you have enough time.\nIn order to make the process faster, I select a samller subsample. ","metadata":{}},{"cell_type":"code","source":"'''from sklearn.model_selection import GridSearchCV\nlgb_train = lgb.Dataset(X_train, label=y_train)\nlgb_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_train)\n\n# Parameter Space\nparam_grid = {\n    'learning_rate': [0.01, 0.05, 0.1]\n}\n\nmodel = lgb.LGBMClassifier(boosting_type='gbdt', objective='binary', metric='auc', n_estimators=1000, verbose=-1)\n\n# GridSearchCV\ngrid_search = GridSearchCV(model, param_grid, cv=3, scoring='roc_auc', verbose=1)\ngrid_search.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], eval_metric='auc')\n\n# Output\nprint(\"Best parameters:\", grid_search.best_params_)\nprint(\"Best score:\", grid_search.best_score_)'''\n#Best parameters: {'learning_rate': 0.05}\n#Best score: 0.7812320159347513","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following cross validation is faster. We try it on optimizing max_depth.","metadata":{}},{"cell_type":"code","source":"'''from sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.model_selection import GridSearchCV\n\n#randomly select 20% observations as sample\nX_sample, _, y_sample, _ = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\nlgb_sample_train = lgb.Dataset(X_sample, label=y_sample)\nlgb_sample_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_sample_train)\n#Parameter Space\nparam_grid = {\n    'max_depth': [3, 5, 7]  \n}\n\nmodel = lgb.LGBMClassifier(boosting_type='gbdt', objective='binary', metric='auc', n_estimators=1000, verbose=-1)\n\n# GridSearchCV\ngrid_search = GridSearchCV(model, param_grid, cv=2, scoring='roc_auc', verbose=1)\ngrid_search.fit(X_sample, y_sample, eval_set=[(X_valid, y_valid)], eval_metric='auc')\n\n# Output\nprint(\"Best parameters:\", grid_search.best_params_)\nprint(\"Best score:\", grid_search.best_score_)'''\n#Best parameters: {'max_depth': 3}\n#Best score: 0.797032247392043","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''from sklearn.model_selection import GridSearchCV\n#randomly select 20% observations as sample\nX_sample, _, y_sample, _ = train_test_split(X_train, y_train, test_size=0.01, random_state=42)\nX_sample_valid, _=train_test_split(X_valid, test_size=0.05, random_state=42)\nlgb_sample_train = lgb.Dataset(X_sample, label=y_sample)\nlgb_sample_valid = lgb.Dataset(X_sample_valid, label=y_valid, reference=lgb_sample_train)\n#Parameter Space\n#param_grid = {'num_leaves':[20,30,40]  }\nparam_grid = {'num_leaves':[29,30,31]  }\nmodel = lgb.LGBMClassifier(boosting_type='gbdt', objective='binary', metric='accuracy', n_estimators=1000, verbose=-1)\n\n# GridSearchCV\ngrid_search = GridSearchCV(model, param_grid, cv=2, scoring='accuracy', verbose=1)\ngrid_search.fit(X_sample, y_sample, eval_set=[(X_valid, y_valid)], eval_metric='accuracy')\n\nprint(\"Best parameters:\", grid_search.best_params_)\nprint(\"Best score:\", grid_search.best_score_)\n'''\n#Anyway, we should combine all hyperparameters together so that we can make a resonable selection of hyperparameters group.\n#And we can also find that using accuracy is not a good choice(it excludes FPR) even though it may be faster.\n#Best parameters: {'num_leaves': 30}\n#Best score: 0.9681398370078439\n#Best parameters: {'num_leaves': 30}\n#Best score: 0.968255526052684","metadata":{"execution":{"iopub.status.busy":"2024-03-10T02:52:16.716687Z","iopub.execute_input":"2024-03-10T02:52:16.717463Z","iopub.status.idle":"2024-03-10T03:31:22.788590Z","shell.execute_reply.started":"2024-03-10T02:52:16.717406Z","shell.execute_reply":"2024-03-10T03:31:22.786238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Record AUC score of each iteration so that we can make a plot to see how iteration improves the ROC(True positive rate vs False postivie rate).\n### Anyway, when we try to add roc_auc_score function in iteration it decreases the running speed of the program significantly. The reason is that we calculate AUC score for twice in order to record them.","metadata":{}},{"cell_type":"code","source":"auc_values = []\n\ndef record_auc(preds, train_data):\n    labels = train_data.get_label()\n    auc = roc_auc_score(labels, preds)\n    auc_values.append(auc)\n    return 'auc', auc, True\ngbm_1 = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,  \n    callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)],\n    feval=record_auc\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(6, 4))\nplt.plot(range(1, len(auc_values) + 1), auc_values, marker='o', color='b', label='AUC values')\nplt.xlabel('Iteration')\nplt.ylabel('AUC')\nplt.title('AUC values over iterations')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### This chart is better.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6, 4))\nplt.plot(range(1, len(auc_values) + 1), auc_values, color='b', linewidth=2, label='AUC values')\nplt.xlabel('Iteration', fontsize=12)\nplt.ylabel('AUC', fontsize=12)\nplt.title('AUC values over iterations', fontsize=14)\nplt.legend(fontsize=12)\nplt.grid(True)\nplt.xticks(fontsize=10)\nplt.yticks(fontsize=10)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Although we only input 1 AUC score, we still can get ROC curve because `metrics.roc_curve()` can get TPR and FPR under different threshold values.\ny_pred = gbm.predict(X_valid)\nauc = metrics.roc_auc_score(y_valid, y_pred)\nprint(f\"AUC: {auc}\")\nfpr, tpr, thresholds = metrics.roc_curve(y_valid, y_pred)\nplt.plot(fpr, tpr, label='ROC curve (area = %0.2f)' % auc)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Present AUC Score of train se,valid set and test set\nfor base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    y_pred = gbm.predict(X, num_iteration=gbm.best_iteration)\n    base[\"score\"] = y_pred\n'''\nBefore tuning (learning_rate=0.01,num_leaves=25)\nThe AUC score on the train set is: 0.79268795973734\nThe AUC score on the valid set is: 0.7879934685792691\nThe AUC score on the test set is: 0.787317029797876\n\nAfter tuning (learning_rate=0.05,num_leaves=30)\nThe AUC score on the train set is: 0.8151541104325495\nThe AUC score on the valid set is: 0.802110080703248\nThe AUC score on the test set is: 0.8006882767326519\n'''\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}') \nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}') \nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}')  ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:46:03.022254Z","iopub.execute_input":"2024-03-10T03:46:03.023136Z","iopub.status.idle":"2024-03-10T03:46:38.507010Z","shell.execute_reply.started":"2024-03-10T03:46:03.023091Z","shell.execute_reply":"2024-03-10T03:46:38.505504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Define gini stability metric**","metadata":{}},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    # from base DataFrame select \"WEEK_NUM\", \"target\",\"score\"\n    selected_columns = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\n\n    # order by \"WEEK_NUM\"\n    sorted_data = selected_columns.sort_values(\"WEEK_NUM\")\n\n    # group by \"WEEK_NUM\"\n    grouped_data = sorted_data.groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\n\n    # Calculate mean of gini, residual of gini\n    gini_in_time = grouped_data.apply(lambda x: 2 * roc_auc_score(x[\"target\"], x[\"score\"]) - 1).tolist()\n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    # linear regression model generates falling rate,namely a, and intercept b\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:01:29.544733Z","iopub.execute_input":"2024-03-09T13:01:29.545162Z","iopub.status.idle":"2024-03-09T13:01:29.555642Z","shell.execute_reply.started":"2024-03-09T13:01:29.545130Z","shell.execute_reply":"2024-03-09T13:01:29.554093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}')  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n### 'data' set is split into training data and testing data.\n### 'data_submission' is consituted by test set and it is used for prediction.\ncols_pred is column names.","metadata":{}},{"cell_type":"code","source":"X_submission = data_submission[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)\n#Here we pick categorical variables' column names\ncategorical_cols = X_train.select_dtypes(include=['category']).columns\n\n#Be careful!!! Here we meet a question:\n'''\nValueError: train and valid dataset categorical_feature do not match.\nIn order to deal with it, we have to check which types are different bewteen sets.\n'''\ncategorical_cols_1 = X_submission.select_dtypes(include=['category']).columns\ntype_difference_1=set(categorical_cols)-set(categorical_cols_1)\ncategorical_cols_2 = X_valid.select_dtypes(include=['category']).columns\ntype_difference_2=set(categorical_cols)-set(categorical_cols_2)\ntype_difference_3=set(categorical_cols_1)-set(categorical_cols)\nprint('Difference in train set and X_submission is',type_difference_1)\nprint('Difference in X_submission set and train set is',type_difference_3)\nprint('Difference in train set and valid set is',type_difference_2)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:02:09.709517Z","iopub.execute_input":"2024-03-09T13:02:09.709992Z","iopub.status.idle":"2024-03-09T13:02:09.888327Z","shell.execute_reply.started":"2024-03-09T13:02:09.709956Z","shell.execute_reply":"2024-03-09T13:02:09.886472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Deal with categorical variables:new type will be regarded as 'unknown \nfor col in categorical_cols:\n    train_categories = set(X_train[col].cat.categories)\n    submission_categories = set(X_submission[col].astype('category').cat.categories)\n    new_categories = submission_categories - train_categories\n    X_submission.loc[X_submission[col].isin(new_categories), col] = \"Unknown\"\n    new_dtype = pd.CategoricalDtype(categories=train_categories, ordered=True)\n    X_train[col] = X_train[col].astype(new_dtype)\n    X_submission[col] = X_submission[col].astype(new_dtype)\n#As you can see,that's the problem. Those categorical variables should be floating variables.\n#Anyway,the convert_string function does not include these categorical variables ending with 'L' in submission set.\n#So we have to convert them into floating variables here.\ndisplay(X_train['fortoday_1092L'].dtypes,X_submission['fortoday_1092L'].dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:02:14.504789Z","iopub.execute_input":"2024-03-09T13:02:14.505653Z","iopub.status.idle":"2024-03-09T13:02:14.848993Z","shell.execute_reply.started":"2024-03-09T13:02:14.505548Z","shell.execute_reply":"2024-03-09T13:02:14.847210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_difference_3_list = list(type_difference_3)\nfor col in type_difference_3_list:\n    X_submission[col].replace('Unknown', np.nan, inplace=True)\n    X_submission[col] = pd.to_numeric(X_submission[col])\nprint(X_submission[type_difference_3_list].dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:02:18.437691Z","iopub.execute_input":"2024-03-09T13:02:18.438090Z","iopub.status.idle":"2024-03-09T13:02:18.538361Z","shell.execute_reply.started":"2024-03-09T13:02:18.438061Z","shell.execute_reply":"2024-03-09T13:02:18.536360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get the prediction of debt default on testing set\n#You can get the probabilty of being positive.\ny_submission_pred_LGB = gbm.predict(X_submission, num_iteration=gbm.best_iteration)\ny_submission_pred_LGB","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score_LGB\": y_submission_pred_LGB\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB\nRegularization:Lasso/Ridge\n\nLoss function\n\nGBDT","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)\ndtrain = xgb.DMatrix(X_train, label=y_train, enable_categorical=True)\ndvalid = xgb.DMatrix(X_valid, label=y_valid, enable_categorical=True)\n\n\nparams = {\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"auc\",\n    \"max_depth\": 5,\n    \"eta\": 0.1,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.8,  \n}\n\n#XGBoost\nXGB = xgb.train(\n    params,\n    dtrain,\n    evals=[(dvalid, \"validation\")],\n    early_stopping_rounds=10,\n    verbose_eval=False  # Close the output of iteration because it is too long\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T07:04:27.415793Z","iopub.execute_input":"2024-03-10T07:04:27.416330Z","iopub.status.idle":"2024-03-10T07:05:45.896465Z","shell.execute_reply.started":"2024-03-10T07:04:27.416293Z","shell.execute_reply":"2024-03-10T07:05:45.895394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_sample, _, y_sample, _ = train_test_split(X_train, y_train, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:58:46.640374Z","iopub.execute_input":"2024-03-10T03:58:46.641221Z","iopub.status.idle":"2024-03-10T03:58:48.670046Z","shell.execute_reply.started":"2024-03-10T03:58:46.641171Z","shell.execute_reply":"2024-03-10T03:58:48.668510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. `max_depth`: `max_depth` is a hyperparameter in the XGBoost model, representing the maximum depth of each tree. It determines the complexity of the tree structure, where a larger `max_depth` can lead to overfitting, while a smaller `max_depth` limits the complexity of the model. In this example, the values for `max_depth` are 3, 5, and 7.\n\n2. `eta`: `eta` is the learning rate parameter in the XGBoost model, also known as the step size. It controls the magnitude of weight adjustment for each tree and a smaller learning rate can stabilize the learning process but may require more iterations. In this example, the values for `eta` are 0.01, 0.05, and 0.1.\n\n3. `colsample_bytree`: `colsample_bytree` is the proportion of features sampled when training each tree in the XGBoost model. This parameter helps prevent overfitting by randomly selecting a subset of features for each tree. In this example, the values for `colsample_bytree` are 0.8 and 0.9, indicating that 80% or 90% of features are randomly selected for training each tree.","metadata":{}},{"cell_type":"code","source":"'''import xgboost as xgb\nfrom sklearn.model_selection import GridSearchCV\nfrom xgboost import XGBClassifier\nxgb_clf = XGBClassifier(objective=\"binary:logistic\", eval_metric=\"auc\", enable_categorical=True)\n# Parameter Space\nparam_grid = {\n    \"max_depth\": [3, 5, 7],\n    \"eta\": [0.01, 0.05, 0.1],\n    \"colsample_bytree\": [0.8, 0.9]\n}\n\ngrid_search = GridSearchCV(estimator=xgb_clf, param_grid=param_grid, scoring=\"roc_auc\", cv=3)\ngrid_search.fit(X_sample, y_sample)\nprint(\"Best parameters found: \", grid_search.best_params_)\nprint(\"Best AUC score found: \", grid_search.best_score_)'''\n#Best parameters found:  {'colsample_bytree': 0.8, 'eta': 0.1, 'max_depth': 5}\n#Best AUC score found:  0.7840095629294401","metadata":{"execution":{"iopub.status.busy":"2024-03-10T03:58:50.858924Z","iopub.execute_input":"2024-03-10T03:58:50.859782Z","iopub.status.idle":"2024-03-10T05:03:50.974795Z","shell.execute_reply.started":"2024-03-10T03:58:50.859734Z","shell.execute_reply":"2024-03-10T05:03:50.973379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BayesSearchCV\nBayesian Optimization method will make use of previous parameters and performance data.\nIt will explore areas in parameter space in case of missing optimal solutions.\nIt use Gaussian Process to guide the parameter searching process. UCB（Upper Confidence Bound）/EI（Expected Improvement） stragety will be applied for selecting next parameter group.\n\nGaussian Process Model\n\nIt takes 4 hours to obtain the result when n_iter = 20.","metadata":{}},{"cell_type":"code","source":"'''import xgboost as xgb\nfrom skopt import BayesSearchCV\nfrom xgboost import XGBClassifier\n\nxgb_clf = XGBClassifier(objective=\"binary:logistic\", eval_metric=\"auc\", enable_categorical=True)\n\n# Parameter Space for Bayesian Optimization\nparam_space = {\n    \"max_depth\": (3, 7),  \n    \"eta\": (0.01, 0.1),\n    \"colsample_bytree\": (0.8, 0.9)\n}\n\nbayes_search = BayesSearchCV(\n    estimator=xgb_clf,\n    search_spaces=param_space,\n    scoring=\"roc_auc\",\n    cv=3,\n    n_iter=20,  # iteration rounds\n    random_state=42\n)\n#AttributeError: module 'numpy' has no attribute 'int'(Use follwing code if you meet this error)\nnp.int = np.int32\nnp.float = np.float64\nnp.bool = np.bool_\n\nbayes_search.fit(X_sample, y_sample)\n\nprint(\"Best parameters found: \", bayes_search.best_params_)\nprint(\"Best AUC score found: \", bayes_search.best_score_)'''\n#Best parameters found:  OrderedDict([('colsample_bytree', 0.8), ('eta', 0.08478857429078285), ('max_depth', 5)])\n#Best AUC score found:  0.7828268723192","metadata":{"execution":{"iopub.status.busy":"2024-03-10T05:25:05.113490Z","iopub.execute_input":"2024-03-10T05:25:05.113946Z","iopub.status.idle":"2024-03-10T06:40:32.637825Z","shell.execute_reply.started":"2024-03-10T05:25:05.113914Z","shell.execute_reply":"2024-03-10T06:40:32.636514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"AUC score before tuning {'colsample_bytree': 0.8, 'eta': 0.05, 'max_depth': 3}\n\nThe AUC score on the train set is: 0.6996403249633859\n\nThe AUC score on the valid set is: 0.6856023486604439\n\nThe AUC score on the test set is: 0.68784401659456\n\n\nAUC score After Bayes Optimization tuning {'colsample_bytree': 0.8, 'eta': 0.08478857429078285, 'max_depth': 5}\n\nThe AUC score on the train set is: 0.7375748880864214\n\nThe AUC score on the valid set is: 0.7156459492720252\n\nThe AUC score on the test set is: 0.7181292349669418\n\nAUC score After Cross Validation tuning {'colsample_bytree': 0.8, 'eta': 0.1, 'max_depth': 5}\n\nThe AUC score on the train set is: 0.7408469963158639\n\nThe AUC score on the valid set is: 0.721816247735543\n\nThe AUC score on the test set is: 0.7241079023146265","metadata":{}},{"cell_type":"code","source":"y_pred_train = XGB.predict(dtrain)\ny_pred_valid = XGB.predict(dvalid)\nfor base, X, y in [(base_train, X_train, y_train), (base_valid, X_valid, y_valid), (base_test, X_test, y_test)]:\n    dmatrix = xgb.DMatrix(X, enable_categorical=True)\n    y_pred = XGB.predict(dmatrix)\n    base[\"score\"] = y_pred\n\nprint(f'The AUC score on the train set is: {roc_auc_score(y_train, y_pred_train)}')\nprint(f'The AUC score on the valid set is: {roc_auc_score(y_valid, y_pred_valid)}')\nprint(f'The AUC score on the test set is: {roc_auc_score(y_test, base_test[\"score\"])}')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T07:06:00.414175Z","iopub.execute_input":"2024-03-10T07:06:00.414654Z","iopub.status.idle":"2024-03-10T07:06:48.560315Z","shell.execute_reply.started":"2024-03-10T07:06:00.414618Z","shell.execute_reply":"2024-03-10T07:06:48.558750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}')  ","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:33:51.976006Z","iopub.execute_input":"2024-03-09T13:33:51.976435Z","iopub.status.idle":"2024-03-09T13:33:53.160064Z","shell.execute_reply.started":"2024-03-09T13:33:51.976398Z","shell.execute_reply":"2024-03-09T13:33:53.158486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtest = xgb.DMatrix(X_submission, enable_categorical=True)\ny_submission_pred_XGB = XGB.predict(dtest)\nprint(y_submission_pred_XGB)","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:36:24.842790Z","iopub.execute_input":"2024-03-09T13:36:24.843189Z","iopub.status.idle":"2024-03-09T13:36:24.911871Z","shell.execute_reply.started":"2024-03-09T13:36:24.843156Z","shell.execute_reply":"2024-03-09T13:36:24.910632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score_XGB\": y_submission_pred_XGB\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-09T13:36:33.910859Z","iopub.execute_input":"2024-03-09T13:36:33.911251Z","iopub.status.idle":"2024-03-09T13:36:33.920026Z","shell.execute_reply.started":"2024-03-09T13:36:33.911221Z","shell.execute_reply":"2024-03-09T13:36:33.918639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Why lgb is better?\n(1)lgb can deal with categorical types while xgb has to use one-hot coding which may increase time and complexity of the model unnecessarily.\n\n(2)histogram-esquealgorithm can save calculation time.\n\n(3)The leaf-wise algorithm reduces more loss when calculating, thus achieving better accuracy compared to the level-wise algorithm.It is more greedy because it chooses the leaf to split the max delta loss while xgboost using level-wise method chooses the best level(it may unnecessarily split some leaves).","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}