{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8288067,"sourceType":"datasetVersion","datasetId":4922958}],"dockerImageVersionId":30674,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport gc # 垃圾回收\nimport sys\nimport subprocess\nfrom datetime import datetime\n\nfrom glob import glob # 从目录中获取文件列表\nfrom pathlib import Path\nfrom itertools import combinations, permutations\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport missingno as msno\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# from sklearn.base import BaseEstimator, RegressorMixin\n# from sklearn.metrics import roc_auc_score\n# from sklearn.preprocessing import OrdinalEncoder, LabelEncoder\n# from sklearn.impute import KNNImputer\n# from sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\n\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import clone\n\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cab","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:50.102821Z","iopub.execute_input":"2024-05-05T14:42:50.103207Z","iopub.status.idle":"2024-05-05T14:42:55.087941Z","shell.execute_reply.started":"2024-05-05T14:42:50.103180Z","shell.execute_reply":"2024-05-05T14:42:55.086723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.数据分析","metadata":{}},{"cell_type":"markdown","source":"# 1.1 导入路径","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.089815Z","iopub.execute_input":"2024-05-05T14:42:55.090375Z","iopub.status.idle":"2024-05-05T14:42:55.096697Z","shell.execute_reply.started":"2024-05-05T14:42:55.090341Z","shell.execute_reply":"2024-05-05T14:42:55.095491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.2 研究train_base数据","metadata":{}},{"cell_type":"code","source":"# df_train_base = pd.read_parquet(TRAIN_DIR / \"train_base.parquet\")\n# print(df_train_base.shape)\n# df_train_base.head(10)      ","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.097809Z","iopub.execute_input":"2024-05-05T14:42:55.098178Z","iopub.status.idle":"2024-05-05T14:42:55.110760Z","shell.execute_reply.started":"2024-05-05T14:42:55.098142Z","shell.execute_reply":"2024-05-05T14:42:55.109047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # case_id 没有重复行\n# # df_train_base.loc[:,'case_id'].duplicated()\n# df_train_base[df_train_base.loc[:,'case_id'].duplicated()]","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.113633Z","iopub.execute_input":"2024-05-05T14:42:55.113992Z","iopub.status.idle":"2024-05-05T14:42:55.123709Z","shell.execute_reply.started":"2024-05-05T14:42:55.113966Z","shell.execute_reply":"2024-05-05T14:42:55.122368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2.1 研究train_static数据","metadata":{}},{"cell_type":"code","source":"# # 翻译字典\n# trans = pd.read_csv('/kaggle/input/feature-definitions-translated0/feature_definitions_translated.csv')\n# # print(trans.head())\n# trans_var = trans['Variable'].tolist()\n# trans_description = trans['Description'].tolist()\n# trans_ch = trans['翻译'].tolist()\n# trans_dict = dict(zip(trans_var,trans_ch))\n# print(len(trans_dict))\n\n# # for i, (en, ch) in enumerate(trans_dict.items()):\n# #     print(en, ch)\n# #     if i % 10 == 0:  # 每10个键值对后打印一个换行符\n# #         print()\n# print(list(trans_dict.items())[:10])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.125101Z","iopub.execute_input":"2024-05-05T14:42:55.125477Z","iopub.status.idle":"2024-05-05T14:42:55.134988Z","shell.execute_reply.started":"2024-05-05T14:42:55.125446Z","shell.execute_reply":"2024-05-05T14:42:55.133755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 读取train_static_cb_0\n# dt_tra_sta = pd.read_parquet(TRAIN_DIR / \"train_static_cb_0.parquet\")\n# print(dt_tra_sta.shape)\n# columns_en = dt_tra_sta.columns\n# # print(columns_en)\n# # 将列标签（列名）替换为trans_dict中的中文标签\n# columns_ch = dt_tra_sta.columns.map(trans_dict)\n# # print(columns_ch)\n\n# dt_tra_sta_ch = dt_tra_sta.rename(columns=trans_dict)\n\n# # 计算缺失值百分比\n# missing_percent = round(dt_tra_sta_ch.isnull().sum()/dt_tra_sta_ch.shape[0]*100,2)\n# df_missing_values = pd.DataFrame({'feature_ch':missing_percent.index, 'missing_rate':missing_percent.values}, index=columns_en).sort_values(by='missing_rate',ascending = False)\n# print(df_missing_values)\n# print('missing_rate>90%:', len(df_missing_values[df_missing_values['missing_rate']>90]))\n\n\n# print(msno.matrix(dt_tra_sta))","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.136400Z","iopub.execute_input":"2024-05-05T14:42:55.136830Z","iopub.status.idle":"2024-05-05T14:42:55.151781Z","shell.execute_reply.started":"2024-05-05T14:42:55.136794Z","shell.execute_reply":"2024-05-05T14:42:55.150557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 清洗数据\n# # 对日期列进行合并处理\n# print('total_nums:', len(dt_tra_sta),\n#       '\\nbirthdate_574D_nums:', len(dt_tra_sta)-dt_tra_sta['birthdate_574D'].isnull().sum(), \n#       '\\ndateofbirth_337D_nums:', len(dt_tra_sta)-dt_tra_sta['dateofbirth_337D'].isnull().sum(), \n#       '\\ndateofbirth_342D_nums:', len(dt_tra_sta)-dt_tra_sta['dateofbirth_342D'].isnull().sum())\n# birthdate = dt_tra_sta[['birthdate_574D', 'dateofbirth_337D', 'dateofbirth_342D']]\n# # print(birthdate.head())\n\n# # 相同日期是否可以对应的上\n# count = dt_tra_sta[['birthdate_574D', 'dateofbirth_337D', 'dateofbirth_342D']].isnull().sum(axis = 1)\n# # print('count:', (count == 3).sum()) #40581\n# print('count_missing', round((count == 3).sum()/len(birthdate)*100,2))\n\n# # 填充，构造出生日期列\n# birthdate_con = birthdate['dateofbirth_337D'].fillna(birthdate['dateofbirth_342D']).fillna(birthdate['birthdate_574D'])\n# # print('birthdate_con:', birthdate_con.head())\n\n# print('birthdate_con_isnull:', round(birthdate_con.isnull().sum()/len(birthdate_con)*100,2))","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.153288Z","iopub.execute_input":"2024-05-05T14:42:55.153838Z","iopub.status.idle":"2024-05-05T14:42:55.177696Z","shell.execute_reply.started":"2024-05-05T14:42:55.153798Z","shell.execute_reply":"2024-05-05T14:42:55.176500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 读取train_static_0_0 and train_static_0_1\n\n# # dt_tra_sta_0_0 = read_file(TRAIN_DIR / \"train_static_0_0.parquet\").to_pandas()\n# # dt_tra_sta_0_1 = read_file(TRAIN_DIR / \"train_static_0_1.parquet\").to_pandas()\n# dt_tra_sta_0_0 = pd.read_parquet(TRAIN_DIR / \"train_static_0_0.parquet\")\n# dt_tra_sta_0_1 = pd.read_parquet(TRAIN_DIR / \"train_static_0_1.parquet\")\n# print(dt_tra_sta_0_0.shape)\n# dt_tra_sta_0 = pd.concat([dt_tra_sta_0_0,dt_tra_sta_0_1], axis=0)\n# print(dt_tra_sta_0.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.179386Z","iopub.execute_input":"2024-05-05T14:42:55.179687Z","iopub.status.idle":"2024-05-05T14:42:55.193740Z","shell.execute_reply.started":"2024-05-05T14:42:55.179663Z","shell.execute_reply":"2024-05-05T14:42:55.192658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 查看缺失值\n\n# columns_en = dt_tra_sta_0.columns\n# dt_tra_sta_0_ch = dt_tra_sta_0.rename(columns=trans_dict)\n\n# # 计算缺失值百分比\n# missing_percent = round(dt_tra_sta_0_ch.isnull().sum()/dt_tra_sta_0_ch.shape[0]*100,2)\n# df_missing_values = pd.DataFrame({'feature_ch':missing_percent.index, 'missing_rate':missing_percent.values}, index=columns_en).sort_values(by='missing_rate',ascending = False)\n# df_missing_values\n\n# # 缺失率>90%的特征数量\n# print('missing_rate>90%:',df_missing_values[df_missing_values.missing_rate>90].shape[0])\n# # 特征数量\n# print('total_missing_nums:', len(dt_tra_sta_0))\n\n# msno.matrix(dt_tra_sta_0)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.195100Z","iopub.execute_input":"2024-05-05T14:42:55.195460Z","iopub.status.idle":"2024-05-05T14:42:55.205339Z","shell.execute_reply.started":"2024-05-05T14:42:55.195434Z","shell.execute_reply":"2024-05-05T14:42:55.204218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 清除垃圾\n# del trans, dt_tra_sta, dt_tra_sta_0_0, dt_tra_sta_0_1\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.208275Z","iopub.execute_input":"2024-05-05T14:42:55.208918Z","iopub.status.idle":"2024-05-05T14:42:55.220420Z","shell.execute_reply.started":"2024-05-05T14:42:55.208890Z","shell.execute_reply":"2024-05-05T14:42:55.219327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 对所有表特征进行翻译\n# table_names = [\"train_base.parquet\", \n               \n#                \"train_static_cb_0.parquet\", \n#                \"train_static_0_1.parquet\", \n\n#                \"train_applprev_1_0.parquet\",\n#                \"train_credit_bureau_a_1_0.parquet\", \n#                \"train_tax_registry_a_1.parquet\", \n#                \"train_tax_registry_b_1.parquet\", \n#                \"train_tax_registry_c_1.parquet\", \n#                \"train_credit_bureau_b_1.parquet\", \n#                \"train_other_1.parquet\", \n#                \"train_person_1.parquet\", \n#                \"train_deposit_1.parquet\", \n#                \"train_debitcard_1.parquet\", \n\n#                \"train_credit_bureau_a_2_0.parquet\", \n#                \"train_credit_bureau_b_2.parquet\", \n#                \"train_applprev_2.parquet\"]\n# for i in table_names:\n#     print(i)\n#     dt = pd.read_parquet(TRAIN_DIR / i)\n#     columns_en = dt.columns\n#     columns_ch = dt.columns.map(trans_dict)\n#     # print(columns_ch)\n#     dt_ch = dt.rename(columns=trans_dict)\n\n#     # 计算缺失值百分比\n#     missing_percent = round(dt_ch.isnull().sum()/dt_ch.shape[0]*100,2)\n#     df_missing_values = pd.DataFrame({'feature_ch':missing_percent.index, 'missing_rate':missing_percent.values}, index=columns_en).sort_values(by='missing_rate',ascending = False)\n#     print(df_missing_values)\n#     print('missing_rate>90%:', len(df_missing_values[df_missing_values['missing_rate']>90]))\n#     # print(msno.matrix(dt))","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.221842Z","iopub.execute_input":"2024-05-05T14:42:55.222382Z","iopub.status.idle":"2024-05-05T14:42:55.232160Z","shell.execute_reply.started":"2024-05-05T14:42:55.222352Z","shell.execute_reply":"2024-05-05T14:42:55.231124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.EDA_all_tables","metadata":{}},{"cell_type":"code","source":"# 读取数据\ndef read_file(path, depth=None):\n    \"\"\"\n    使用的是pandas 读取文件\n    \"\"\"\n    print(f\"------正在处理{path}--------\")\n    df = pd.read_parquet(path)\n    df = reduce_mem_usage(df)\n    if depth in [1,2]:\n        df = generate_agg_feature(df)\n    return df\n\ndef read_files(regex_path, depth=None):\n    \"\"\"\n    使用的是pandas 读取文件\n    \"\"\"\n    print(f\"------正在处理{regex_path}--------\")\n    \n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pd.read_parquet(path)\n        chunks.append(df)\n    df = pd.concat(chunks, axis=0)\n    df = reduce_mem_usage(df)\n    if depth in [1,2]:\n        df = generate_agg_feature(df)\n    # print(df['opencred_647L'])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.233183Z","iopub.execute_input":"2024-05-05T14:42:55.234570Z","iopub.status.idle":"2024-05-05T14:42:55.247939Z","shell.execute_reply.started":"2024-05-05T14:42:55.234526Z","shell.execute_reply":"2024-05-05T14:42:55.246678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 减少内存\n\ndef reduce_mem_usage(df, slient=True):\n    \"\"\"\n    减少df的内存使用，并且将类型进行转换\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n\n    if not slient:\n        print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n            df[col] = df[col].astype(np.int64)\n        elif col in [\"date_decision\"]:\n            df[col] = pd.to_datetime(df[col])\n        elif col[-1] in [\"D\"]:\n            df[col] = pd.to_datetime(df[col])\n        elif col[-1] in (\"P\", \"A\"):\n            df[col] = df[col].astype(np.float64)\n\n\n        col_type = df[col].dtype\n        if col_type in [\"category\", \"<M8[ns]\", \"datetime64[ns]\"]:\n            continue\n\n        if col_type not in [object, bool]: # 说明是浮点类型或者整数\n\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else: # 将bool和object类型转化为category\n \n            df[col] = df[col].astype('category')\n\n\n    if not slient:\n        end_mem = df.memory_usage().sum() / 1024**2\n        print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n        print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.249621Z","iopub.execute_input":"2024-05-05T14:42:55.249987Z","iopub.status.idle":"2024-05-05T14:42:55.267077Z","shell.execute_reply.started":"2024-05-05T14:42:55.249960Z","shell.execute_reply":"2024-05-05T14:42:55.265847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 转换为未指定变换\ndef rename_column_name(s):\n    if 'nunique_' in s and s[-1] == \"D\": # 唯一转换时间\n        return s[:-1] + \"L\"\n    return s","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.268528Z","iopub.execute_input":"2024-05-05T14:42:55.269521Z","iopub.status.idle":"2024-05-05T14:42:55.280985Z","shell.execute_reply.started":"2024-05-05T14:42:55.269486Z","shell.execute_reply":"2024-05-05T14:42:55.279933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特征工程\ndef generate_agg_feature(df):\n    \"\"\"\n    生成聚合特征\n    \"\"\"\n    datetime64_columns = df.select_dtypes(\"datetime64[ns]\").columns\n    interger_columns = df.select_dtypes(exclude=[\"datetime64[ns]\", \"category\"]).columns\n    category_columns = df.select_dtypes(\"category\").columns\n\n    datetime64_columns = [x for x in datetime64_columns if x != \"case_id\"]\n    interger_columns = [x for x in interger_columns if x != \"case_id\"]\n    category_columns = [x for x in category_columns if x != \"case_id\"]\n\n    agg_formual = {}\n    for col in df.columns:\n        if col in [\"case_id\"]:\n            continue\n        if col in interger_columns:\n            agg_formual[col] = [\n                \"max\", \"min\", \n                \"mean\",\n                \"nunique\", \"sum\"\n            ]\n        if col in datetime64_columns:\n            agg_formual[col] = [\n                \"max\", \n                \"min\", \n                'first',\n                'last',\n                \"nunique\"\n            ]\n        if col in category_columns:\n            agg_formual[col] = [\n                \"nunique\",\n#                 \"min\",\n#                 \"max\",\n                \"first\",\n                \"last\"\n#                 \"mean\"\n            ]\n\n    df_feat = df.groupby(\"case_id\").agg(agg_formual)\n    df_feat.columns = [x[1] + '_' + x[0] for x in df_feat.columns]\n    df_feat.columns = [rename_column_name(x) for x in df_feat.columns]\n    df_feat = df_feat.reset_index()\n\n    return df_feat\n\n\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    \"\"\"\n    将depth进行merge合并\n    \"\"\"\n    dfs = depth_0 + depth_1 + depth_2\n\n    for i, df in enumerate(dfs):\n        df_base = df_base.merge(df, how=\"left\", on=\"case_id\", suffixes=(\"\", f\"_{i}\"))\n    \n#     df_base = birthdate_con(df_base)\n    df_base = add_month_week(df_base)\n    df_base = handle_dates(df_base)\n#     df_base = filter_cols(df_base)\n#     df_base = sel_cat_cols(df_base)\n    \n    return df_base\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:42:55.282697Z","iopub.execute_input":"2024-05-05T14:42:55.283264Z","iopub.status.idle":"2024-05-05T14:42:55.294095Z","shell.execute_reply.started":"2024-05-05T14:42:55.283235Z","shell.execute_reply":"2024-05-05T14:42:55.293085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def birthdate_con(df):\n    # 填充，构造出生日期列\n    birthdate = df[['birthdate_574D', 'dateofbirth_337D', 'dateofbirth_342D']]\n    df['dateofbirth_337D'] = birthdate['dateofbirth_337D'].fillna(birthdate['dateofbirth_342D']).fillna(birthdate['birthdate_574D'])\n#     df.drop(['birthdate_574D', 'dateofbirth_342D'], inplace = True)\n    return df\n\ndef add_month_week(df):\n    # 构造月份和星期特征\n    df['month_decision'] = df['date_decision'].dt.month\n    df['weekday_decision'] = df['date_decision'].dt.weekday\n    return df\n\ndef handle_dates(df):\n    # 将日期差转化为day\n    for col in df.columns:\n        if col.endswith(\"D\"):\n            # 计算差值\n            df[col] = (df[col] - df[\"date_decision\"]).dt.days\n            # 删除日期列和 \"MONTH\" 列\n    df.drop(columns=[\"date_decision\", \"MONTH\"], inplace=True)\n    return df\n\ndef filter_cols(df):\n    # 缺失值处理 删除缺失值大于90%的特征\n    for col in df.columns:\n        if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n            isnull_ratio = df[col].isnull().mean()\n            if isnull_ratio > 0.9:\n                df.drop(columns=[col], inplace=True)\n\n    # 删除唯一值较多或者只有一个值的字符串列\n    for col in df.columns:\n        if col not in [\"target\", \"case_id\", \"WEEK_NUM\"] and df[col].dtype == 'object':\n            unique_count = df[col].nunique()\n            if unique_count == 1 or unique_count > 200:\n                df.drop(columns=[col], inplace=True)\n    return df\n\ndef sel_cat_cols(df, cat_cols=None):\n#     df = df.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df.select_dtypes(\"object\").columns)\n    df[cat_cols] = df[cat_cols].astype(\"category\")\n    return df, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:55:25.816958Z","iopub.execute_input":"2024-05-05T14:55:25.817357Z","iopub.status.idle":"2024-05-05T14:55:25.830320Z","shell.execute_reply.started":"2024-05-05T14:55:25.817331Z","shell.execute_reply":"2024-05-05T14:55:25.829193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 降维\ndef feature_reduction(df):\n    nums=df.select_dtypes(exclude='category').columns\n\n    #df_train=df_train[nums]\n    nans_df = df[nums].isna()\n    nans_groups={}\n    for col in nums:\n        cur_group = nans_df[col].sum()\n        try:\n            nans_groups[cur_group].append(col)\n        except:\n            nans_groups[cur_group]=[col]\n    del nans_df; x=gc.collect()\n\n    def reduce_group(grps):\n        use = []\n        for g in grps:\n            mx = 0; vx = g[0]\n            for gg in g:\n                n = df[gg].nunique()\n                if n>mx:\n                    mx = n\n                    vx = gg\n                #print(str(gg)+'-'+str(n),', ',end='')\n            use.append(vx)\n            #print()\n        print('Use these',use)\n        return use\n\n    def group_columns_by_correlation(matrix, threshold=0.8):\n        # 计算列之间的相关性\n        correlation_matrix = matrix.corr()\n\n        # 分组列\n        groups = []\n        remaining_cols = list(matrix.columns)\n        while remaining_cols:\n            col = remaining_cols.pop(0)\n            group = [col]\n            correlated_cols = [col]\n            for c in remaining_cols:\n                if correlation_matrix.loc[col, c] >= threshold:\n                    group.append(c)\n                    correlated_cols.append(c)\n            groups.append(group)\n            remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n\n        return groups\n\n    uses=[]\n    for k,v in nans_groups.items():\n        if len(v)>1:\n                Vs = nans_groups[k]\n                #cross_features=list(combinations(Vs, 2))\n                #make_corr(Vs)\n                grps= group_columns_by_correlation(df[Vs], threshold=0.8)\n                use=reduce_group(grps)\n                uses=uses+use\n                #make_corr(use)\n        else:\n            uses=uses+v\n        print('####### NAN count =',k)\n#     print(uses)\n#     print(len(uses))\n    uses=uses+list(df.select_dtypes(include='category').columns)\n#     print(len(uses))\n    df=df[uses]\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:55:26.222136Z","iopub.execute_input":"2024-05-05T14:55:26.222521Z","iopub.status.idle":"2024-05-05T14:55:26.235165Z","shell.execute_reply.started":"2024-05-05T14:55:26.222492Z","shell.execute_reply":"2024-05-05T14:55:26.233673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 读取数据\n# %%time \n# data_store = {\n#     \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n#         read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n#     ],\n#     \"depth_1\": [\n#         read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n#         read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1), # 新增\n#         read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_0.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_1.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_2.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_3.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_4.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_5.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_6.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_7.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_8.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_9.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_a_2_10.parquet\", 2), # 新增\n#         read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n#         read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_person_2.parquet\", 2) # 新增\n#     ]\n# }\n\n# data_store = {\n#     \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n#         read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n#     ], \n#     \"depth_1\": [\n#         read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n# #         read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n#     ]\n# }\n\ndata_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n    ],\n        \"depth_2\": [\n    ]\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:55:26.762837Z","iopub.execute_input":"2024-05-05T14:55:26.763197Z","iopub.status.idle":"2024-05-05T14:57:28.020461Z","shell.execute_reply.started":"2024-05-05T14:55:26.763172Z","shell.execute_reply":"2024-05-05T14:57:28.019170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_train = feature_eng(**data_store)\n# df_base = data_store['df_base']\ndf_train = filter_cols(df_train)\n# print('df_train.shape:', df_train.shape)\n# df_train.head(20)\ndf_train, cat_cols = sel_cat_cols(df_train)\ndf_train = feature_reduction(df_train)\n\ndel data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-05T14:57:28.022415Z","iopub.execute_input":"2024-05-05T14:57:28.024259Z","iopub.status.idle":"2024-05-05T15:01:19.710697Z","shell.execute_reply.started":"2024-05-05T14:57:28.024223Z","shell.execute_reply":"2024-05-05T15:01:19.709241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data_store_test = {\n#     \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n#         read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n#     ],\n#     \"depth_1\": [\n#         read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n#         read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n#         read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_0.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_1.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_2.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_3.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_4.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_5.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_6.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_7.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_8.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_9.parquet\", 2), # 新增\n#         read_file(TEST_DIR / \"test_credit_bureau_a_2_10.parquet\", 2), # 新增\n# #         read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n#         read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n#         read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n#     ]\n# }\n\n# %%time\n# data_store_test = {\n#     \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n#         read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n#     ],\n#     \"depth_1\": [\n#         read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n#     ]\n# }\n\ndata_store_test = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:01:19.712392Z","iopub.execute_input":"2024-05-05T15:01:19.713225Z","iopub.status.idle":"2024-05-05T15:01:20.219605Z","shell.execute_reply.started":"2024-05-05T15:01:19.713194Z","shell.execute_reply":"2024-05-05T15:01:20.218182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_test = feature_eng(**data_store_test)\n\ndel data_store_test\ngc.collect()\n\n# df_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n# print(\"train data shape:\\t\", df_train.shape)\n# print(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:01:20.222958Z","iopub.execute_input":"2024-05-05T15:01:20.224024Z","iopub.status.idle":"2024-05-05T15:01:20.469307Z","shell.execute_reply.started":"2024-05-05T15:01:20.223990Z","shell.execute_reply":"2024-05-05T15:01:20.467945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"train data shape:\\t\", df_train.shape)\n# print(\"test data shape:\\t\", df_test.shape)\n# k = 0\n# for i in df_train.columns:\n#     for j in df_test.columns:\n#         if i == j:\n#             k += 1\n# print('k:', k)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:01:20.470857Z","iopub.execute_input":"2024-05-05T15:01:20.471196Z","iopub.status.idle":"2024-05-05T15:01:20.476530Z","shell.execute_reply.started":"2024-05-05T15:01:20.471171Z","shell.execute_reply":"2024-05-05T15:01:20.475278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test[[col for col in df_train.columns if col != \"target\"]]\n\ndf_test, cat_cols = sel_cat_cols(df_test, cat_cols)\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:01:20.478116Z","iopub.execute_input":"2024-05-05T15:01:20.478572Z","iopub.status.idle":"2024-05-05T15:01:20.501395Z","shell.execute_reply.started":"2024-05-05T15:01:20.478544Z","shell.execute_reply":"2024-05-05T15:01:20.499877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train is duplicated:\\t\", df_train[\"case_id\"].duplicated().any())\nprint(\"Train Week Range:\\t\", (df_train[\"WEEK_NUM\"].min(), df_train[\"WEEK_NUM\"].max()))\n\nprint()\n\nprint(\"Test is duplicated:\\t\", df_test[\"case_id\"].duplicated().any())\nprint(\"Test Week Range:\\t\", (df_test[\"WEEK_NUM\"].min(), df_test[\"WEEK_NUM\"].max()))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:01:20.502775Z","iopub.execute_input":"2024-05-05T15:01:20.503460Z","iopub.status.idle":"2024-05-05T15:01:20.533687Z","shell.execute_reply.started":"2024-05-05T15:01:20.503428Z","shell.execute_reply":"2024-05-05T15:01:20.532701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(\n    data=df_train,\n    x=\"WEEK_NUM\",\n    y=\"target\",\n)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:01:20.535100Z","iopub.execute_input":"2024-05-05T15:01:20.535741Z","iopub.status.idle":"2024-05-05T15:01:34.324731Z","shell.execute_reply.started":"2024-05-05T15:01:20.535710Z","shell.execute_reply":"2024-05-05T15:01:34.323696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train['description_5085714M'].dtype)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:03:05.482716Z","iopub.execute_input":"2024-05-05T15:03:05.483993Z","iopub.status.idle":"2024-05-05T15:03:05.489783Z","shell.execute_reply.started":"2024-05-05T15:03:05.483956Z","shell.execute_reply":"2024-05-05T15:03:05.488544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'description_5085714M' in cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-05T15:03:53.720819Z","iopub.execute_input":"2024-05-05T15:03:53.721188Z","iopub.status.idle":"2024-05-05T15:03:53.728243Z","shell.execute_reply.started":"2024-05-05T15:03:53.721154Z","shell.execute_reply":"2024-05-05T15:03:53.726993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.构建模型","metadata":{}},{"cell_type":"code","source":"y = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\ndf_train= df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\n# cv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:53.659497Z","iopub.execute_input":"2024-05-05T13:56:53.660276Z","iopub.status.idle":"2024-05-05T13:56:54.287116Z","shell.execute_reply.started":"2024-05-05T13:56:53.660240Z","shell.execute_reply":"2024-05-05T13:56:54.286295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df_train), len(y))","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:54.288226Z","iopub.execute_input":"2024-05-05T13:56:54.288498Z","iopub.status.idle":"2024-05-05T13:56:54.293611Z","shell.execute_reply.started":"2024-05-05T13:56:54.288475Z","shell.execute_reply":"2024-05-05T13:56:54.292592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\ndevice='gpu'\n#n_samples=200000\nn_est=6000\n# DRY_RUN = True if sample.shape[0] == 10 else False   \n# if DRY_RUN:\n#     device='cpu'\n#     df_train = df_train.iloc[:60000]\n#     #n_samples=10000\n#     n_est=600\n# print(device)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:58:26.103611Z","iopub.execute_input":"2024-05-05T13:58:26.104459Z","iopub.status.idle":"2024-05-05T13:58:26.117679Z","shell.execute_reply.started":"2024-05-05T13:58:26.104427Z","shell.execute_reply":"2024-05-05T13:58:26.116763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom catboost import CatBoostClassifier, Pool\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nimport gc\n\n# 定义模型参数\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,  \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,  \n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\": True,\n    'num_leaves': 64,\n    \"device\": device, \n    \"verbose\": -1,\n}\n\nparams2 = {\n    \"booster\": \"gbtree\",\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.05,\n    \"n_estimators\": 1000,\n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"alpha\": 0.1,  \n    \"lambda\": 10,  \n    \"tree_method\": 'gpu_hist' if device == 'gpu' else 'auto',\n    \"random_state\": 42,\n    \"verbosity\": 0,\n    \"enable_categorical\": True,\n}\n\n# 划分数据集\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\ndf_train[cat_cols] = df_train[cat_cols].astype(str)\ndf_test[cat_cols] = df_test[cat_cols].astype(str)\n\n# 定义基础模型\nmodels = [\n    ('CatBoost', CatBoostClassifier(eval_metric='AUC', task_type='GPU', learning_rate=0.03, iterations=n_est, random_seed=3107)),\n    ('LightGBM', LGBMClassifier(**params)),\n    ('XGBoost', XGBClassifier(**params2))\n]\n\n# 定义元模型\nmeta_params = {\n    'n_estimators': 12,\n    'learning_rate': 0.1,\n    'max_depth': 3,\n    'min_samples_split': 3,\n    'min_samples_leaf': 1\n}\n\nmeta_model = GradientBoostingClassifier(**meta_params)\n\n# 存储基础模型和相应的AUC分数\nfitted_models = {'CatBoost': [], 'LightGBM': [], 'XGBoost': []}\ncv_scores = {'CatBoost': [], 'LightGBM': [], 'XGBoost': []}\n\n# 存储第一层模型的预测结果\nmeta_features = pd.DataFrame(index=df_train.index, columns=['CatBoost', 'LightGBM', 'XGBoost'])\n\n# 训练和评估基础模型\nfor name, model in models:\n    for idx_train, idx_valid in cv.split(df_train, y, groups=weeks):\n#         X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]\n#         X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n        X_train, y_train = df_train.iloc[idx_train].copy(), y.iloc[idx_train]\n        X_valid, y_valid = df_train.iloc[idx_valid].copy(), y.iloc[idx_valid]\n\n        if name == 'CatBoost':\n            X_train[cat_cols] = X_train[cat_cols].astype(str)\n            X_valid[cat_cols] = X_valid[cat_cols].astype(str)\n            train_pool = Pool(X_train, y_train, cat_features=cat_cols)\n            val_pool = Pool(X_valid, y_valid, cat_features=cat_cols)\n            model.fit(train_pool, eval_set=val_pool, verbose=False)\n            y_pred_valid = model.predict_proba(val_pool)[:, 1]\n#                         train_pool = Pool(X_train, y_train, cat_features=cat_cols)\n#             val_pool = Pool(X_valid, y_valid, cat_features=cat_cols)\n#             model.fit(train_pool, eval_set=val_pool, verbose=False)\n#             y_pred_valid = model.predict_proba(val_pool)[:, 1]\n        elif name == 'LightGBM':\n            X_train[cat_cols] = X_train[cat_cols].astype('category')\n            X_valid[cat_cols] = X_valid[cat_cols].astype('category')\n            model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], callbacks=[lgb.log_evaluation(200), lgb.early_stopping(100)])\n            y_pred_valid = model.predict_proba(X_valid)[:, 1]\n        else:  # XGBoost\n            X_train[cat_cols] = X_train[cat_cols].astype('category')\n            X_valid[cat_cols] = X_valid[cat_cols].astype('category')\n            model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], early_stopping_rounds=100, verbose=False)\n            y_pred_valid = model.predict_proba(X_valid)[:, 1]\n\n        fitted_models[name].append(model)\n        auc_score = roc_auc_score(y_valid, y_pred_valid)\n        cv_scores[name].append(auc_score)\n        meta_features.loc[X_valid.index, name] = y_pred_valid\n\n# 训练元模型\nmeta_model.fit(meta_features, y)\n\n# 预测测试集\ntest_meta_features = pd.DataFrame(index=df_test.index, columns=['CatBoost', 'LightGBM', 'XGBoost'])\nfor name, models_list in fitted_models.items():\n    for model in models_list:\n        if name == 'CatBoost':\n            df_test[cat_cols] = df_test[cat_cols].astype(str)\n            y_pred_test = model.predict_proba(df_test)[:, 1]\n        elif name == 'LightGBM':\n            df_test[cat_cols] = df_test[cat_cols].astype(\"category\")\n            y_pred_test = model.predict_proba(df_test)[:, 1]\n        else:  # XGBoost\n            df_test[cat_cols] = df_test[cat_cols].astype(\"category\")\n            y_pred_test = model.predict_proba(df_test)[:, 1]\n        test_meta_features[name] = test_meta_features[name].add(y_pred_test, fill_value=0)\n\ntest_meta_features /= 5  # 平均预测结果\n\n# 预测结果存储到 CSV 文件中\ny_pred = pd.Series(meta_model.predict_proba(test_meta_features)[:, 1], index=df_test.index)\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\ndf_subm[\"score\"] = y_pred\ndf_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:58:27.333070Z","iopub.execute_input":"2024-05-05T13:58:27.333734Z","iopub.status.idle":"2024-05-05T13:59:27.939395Z","shell.execute_reply.started":"2024-05-05T13:58:27.333702Z","shell.execute_reply":"2024-05-05T13:59:27.937979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# # StratifiedGroupKFold交叉验证\n# cv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\n# # 构建Stacking模型\n# stacked_predictions = np.zeros((X.shape[0], 2))\n# test_predictions = np.zeros((df_test.shape[0], 2))\n\n# for i, (train_idx, val_idx) in enumerate(cv.split(X, y, groups=weeks)):\n#     X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n#     X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n#     # LightGBM模型\n#     print(f\"\\nTraining LightGBM model for fold {i+1}...\")\n#     lgb_model = lgb.LGBMClassifier(**lgb_params)\n#     lgb_model.fit(X_train, y_train, eval_set=[(X_val, y_val)],\n#                   eval_metric='auc')\n#     lgb_val_pred = lgb_model.predict_proba(X_val, num_iteration=lgb_model.best_iteration_)[:, 1]\n#     stacked_predictions[val_idx, 0] = lgb_val_pred\n#     print(f\"Fold {i+1} LightGBM AUC:\", roc_auc_score(y_val, lgb_val_pred))\n\n#     # CatBoost模型\n#     print(f\"\\nTraining CatBoost model for fold {i+1}...\")\n#     cat_model = CatBoostClassifier(**cat_params)\n#     cat_model.fit(X_train, y_train, eval_set=(X_val, y_val), verbose=100, early_stopping_rounds=100)\n#     cat_val_pred = cat_model.predict_proba(X_val, ntree_end=cat_model.best_iteration_)[:, 1]\n#     stacked_predictions[val_idx, 1] = cat_val_pred\n#     print(f\"Fold {i+1} CatBoost AUC:\", roc_auc_score(y_val, cat_val_pred))\n\n#     test_predictions[:, 0] += lgb_model.predict_proba(df_test, num_iteration=lgb_model.best_iteration_)[:, 1] / cv.n_splits\n#     test_predictions[:, 1] += cat_model.predict_proba(df_test, ntree_end=cat_model.best_iteration_)[:, 1] / cv.n_splits\n\n# # 构建Stacking模型的第二层(使用逻辑回归模型)\n# stacked_model = LogisticRegression()\n# stacked_model.fit(stacked_predictions, y)\n\n# # 输出测试集预测结果\n# test_output_df[\"lgb\"] = test_predictions[:, 0]\n# test_output_df[\"cat\"] = test_predictions[:, 1]\n\n# print(\"\\nTest set predictions:\")\n# print(test_output_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.128216Z","iopub.status.idle":"2024-05-05T13:56:55.128513Z","shell.execute_reply.started":"2024-05-05T13:56:55.128360Z","shell.execute_reply":"2024-05-05T13:56:55.128372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 数据准备\n# X = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\n# y = df_train[\"target\"]\n# weeks = df_train[\"WEEK_NUM\"]\n\n# # StratifiedGroupKFold交叉验证\n# cv = StratifiedGroupKFold(n_splits=5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.130343Z","iopub.status.idle":"2024-05-05T13:56:55.130692Z","shell.execute_reply.started":"2024-05-05T13:56:55.130529Z","shell.execute_reply":"2024-05-05T13:56:55.130543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # LightGBM模型参数调优\n# lgb_params = {\n#     \"boosting_type\": [\"gbdt\", \"dart\"],\n#     \"learning_rate\": [0.05, 0.1, 0.2],\n#     \"num_leaves\": [31, 50, 100],\n#     \"max_depth\": [5, 7, -1],\n#     \"min_child_samples\": [20, 50, 100],\n#     \"subsample\": [0.6, 0.8, 1.0],\n# }\n\n# print(\"Tuning LightGBM model...\")\n# lgb_clf = lgb.LGBMClassifier(objective='binary', metric='auc', random_state=42)\n# lgb_grid = GridSearchCV(estimator=lgb_clf, param_grid=lgb_params, scoring='roc_auc', cv=5)\n# lgb_grid.fit(X, y)\n# print(\"Best LightGBM parameters:\", lgb_grid.best_params_)\n# print(\"Best LightGBM score:\", lgb_grid.best_score_)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.132341Z","iopub.status.idle":"2024-05-05T13:56:55.132671Z","shell.execute_reply.started":"2024-05-05T13:56:55.132500Z","shell.execute_reply":"2024-05-05T13:56:55.132514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # CatBoost模型参数调优\n# cat_params = {\n#     \"learning_rate\": [0.05, 0.1, 0.2],\n#     \"depth\": [4, 6, 8],\n#     \"l2_leaf_reg\": [1, 3, 5],\n# }\n\n# print(\"\\nTuning CatBoost model...\")\n# cat_clf = CatBoostClassifier(task_type=\"GPU\", verbose=0, random_state=42)\n# cat_grid = GridSearchCV(estimator=cat_clf, param_grid=cat_params, scoring='roc_auc', cv=5)\n# cat_grid.fit(X, y)\n# print(\"Best CatBoost parameters:\", cat_grid.best_params_)\n# print(\"Best CatBoost score:\", cat_grid.best_score_)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.134069Z","iopub.status.idle":"2024-05-05T13:56:55.134396Z","shell.execute_reply.started":"2024-05-05T13:56:55.134237Z","shell.execute_reply":"2024-05-05T13:56:55.134253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # 构建Stacking模型\n# stacked_predictions = np.zeros((X.shape[0], 2))\n# test_predictions = np.zeros((df_test.shape[0], 2))\n\n# for i, (train_idx, val_idx) in enumerate(cv.split(X, y, groups=weeks)):\n#     X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n#     X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n#     # LightGBM模型\n#     print(f\"\\nTraining LightGBM model for fold {i+1}...\")\n#     lgb_model = lgb.LGBMClassifier(**lgb_grid.best_params_)\n#     lgb_model.fit(X_train, y_train, eval_set=[(X_val, y_val)], early_stopping_rounds=100, verbose=100)\n#     lgb_val_pred = lgb_model.predict_proba(X_val)[:, 1]\n#     stacked_predictions[val_idx, 0] = lgb_val_pred\n#     print(f\"Fold {i+1} LightGBM AUC:\", roc_auc_score(y_val, lgb_val_pred))\n\n#     # CatBoost模型\n#     print(f\"\\nTraining CatBoost model for fold {i+1}...\")\n#     cat_model = CatBoostClassifier(task_type=\"GPU\", **cat_grid.best_params_, verbose=0, random_state=42)\n#     cat_model.fit(X_train, y_train, eval_set=(X_val, y_val), early_stopping_rounds=100, verbose=100)\n#     cat_val_pred = cat_model.predict_proba(X_val)[:, 1]\n#     stacked_predictions[val_idx, 1] = cat_val_pred\n#     print(f\"Fold {i+1} CatBoost AUC:\", roc_auc_score(y_val, cat_val_pred))\n\n#     test_predictions[:, 0] += lgb_model.predict_proba(df_test)[:, 1] / cv.n_splits\n#     test_predictions[:, 1] += cat_model.predict_proba(df_test)[:, 1] / cv.n_splits\n\n# # 构建Stacking模型\n# final_model = lgb.LGBMClassifier(objective='binary', metric='auc', random_state=42)\n# final_model.fit(stacked_predictions, y)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.135680Z","iopub.status.idle":"2024-05-05T13:56:55.136135Z","shell.execute_reply.started":"2024-05-05T13:56:55.135884Z","shell.execute_reply":"2024-05-05T13:56:55.135902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 输出测试集预测结果\n# test_output_df[\"lgb\"] = test_predictions[:, 0]\n# test_output_df[\"cat\"] = test_predictions[:, 1]\n# test_output_df[\"stacked_prediction\"] = final_model.predict(test_predictions)\n\n# print(\"Test set predictions:\")\n# print(test_output_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.137820Z","iopub.status.idle":"2024-05-05T13:56:55.138326Z","shell.execute_reply.started":"2024-05-05T13:56:55.138066Z","shell.execute_reply":"2024-05-05T13:56:55.138085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_test = df_test.drop(columns=[\"WEEK_NUM\"])\n# X_test = X_test.set_index(\"case_id\")\n\n\n# y_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)\n\n# df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\n# df_subm = df_subm.set_index(\"case_id\")\n\n# df_subm[\"score\"] = y_pred\n\n# print(\"Check null: \", df_subm[\"score\"].isnull().any())\n# df_subm.head()\n\n# df_subm.to_csv(\"submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.143919Z","iopub.status.idle":"2024-05-05T13:56:55.144289Z","shell.execute_reply.started":"2024-05-05T13:56:55.144123Z","shell.execute_reply":"2024-05-05T13:56:55.144137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_predictions_stacked = stacked_model.predict_proba(test_predictions)[:, 1]\n# submission_df = pd.DataFrame(test_predictions_stacked, index=df_test[\"case_id\"], columns=[\"score\"])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.145472Z","iopub.status.idle":"2024-05-05T13:56:55.145761Z","shell.execute_reply.started":"2024-05-05T13:56:55.145617Z","shell.execute_reply":"2024-05-05T13:56:55.145629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"Check null: \", df_subm[\"score\"].isnull().any())\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.146878Z","iopub.status.idle":"2024-05-05T13:56:55.147205Z","shell.execute_reply.started":"2024-05-05T13:56:55.147049Z","shell.execute_reply":"2024-05-05T13:56:55.147062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.148051Z","iopub.status.idle":"2024-05-05T13:56:55.148355Z","shell.execute_reply.started":"2024-05-05T13:56:55.148204Z","shell.execute_reply":"2024-05-05T13:56:55.148217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T13:56:55.149907Z","iopub.status.idle":"2024-05-05T13:56:55.150392Z","shell.execute_reply.started":"2024-05-05T13:56:55.150152Z","shell.execute_reply":"2024-05-05T13:56:55.150171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}