{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":9726245,"sourceType":"datasetVersion","datasetId":5951490},{"sourceId":146945,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":124660,"modelId":147688},{"sourceId":146987,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":124700,"modelId":147725}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport joblib\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\n# import seaborn as sns\n# import matplotlib.pyplot as plt\n\n# from sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\n# from sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')\nprint(pl.__version__)\n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\nclass Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if (col[-1] in (\"D\",)) and ('count' not in col):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            # if col in [\"decision_month\", \"decision_weekday\"]:\n            #     df = df.drop(col)\n            #     continue\n            # if ('amtde' in col) or ('bureau_b2' in col): # for ohter option\n            #     continue\n\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            # if '_depth2_' in col:\n            #     continue\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\", ]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 50):#50 #len(df) * 0.20): # 95 # fe4 down at fq20\n                    df = df.drop(col)\n            \n            # eliminate yaer, month feature\n            # 644\n            if (col[-1] not in [\"P\", \"A\", \"L\", \"M\"]) and (('month_' in col) or ('year_' in col)):# or ('num_group' in col):\n            # if (('month_' in col) or ('year_' in col)):# or ('num_group' in col):\n                df = df.drop(col)\n\n        return df\n# 644 lb best\nclass Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_2 = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        # expr_3 = [pl.median(col).alias(f\"median_{col}\") for col in cols]\n        # expr_3 = [pl.var(col).alias(f\"var_{col}\") for col in cols]+ [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        # expr_3 = [pl.last(col).alias(f\"last_{col}\") for col in cols] #+ \\\n        #     [pl.first(col).alias(f\"first_{col}\") for col in cols] + \\\n        #     [pl.mean(col).alias(f\"mean_{col}\") for col in cols] + \\\n        #     [pl.std(col).alias(f\"std_{col}\") for col in cols]\n        # expr_3 = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n\n        cols2 = [col for col in df.columns if col[-1] in (\"L\", \"A\")]\n        expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2] + [pl.std(col).alias(f\"std_{col}\") for col in cols2] + \\\n            [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] + [pl.median(col).alias(f\"median_{col}\") for col in cols2] # + \\\n            # [pl.first(col).alias(f\"first_{col}\") for col in cols2] + [pl.last(col).alias(f\"last_{col}\") for col in cols2]\n        \n        # BAD\n        # cols3 = [col for col in df.columns if col[-1] in (\"A\")]\n        # expr_4 = [pl.col(col).fill_null(strategy=\"zero\").apply(lambda x: x.max() - x.min()).alias(f\"max-min_gap_{col}\") \n        #           for col in cols3]\n        return expr_1 + expr_2 + expr_3 # + [pl.col(col).diff().last().alias(f\"diff-last_{col}\") for col in cols3] # + expr_4\n    \n    @staticmethod\n    def applprev2_exprs(df):\n        cols = [col for col in df.columns if \"num_group\" not in col]\n        # expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] \n        expr_2 = [pl.first(col).alias(f\"first_{col}\") for col in cols]#  + [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        return []#expr_2\n\n    @staticmethod\n    def bureau_a1(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_2 = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        cols2 = [\n            # bad\n        'annualeffectiverate_199L', 'annualeffectiverate_63L',\n        'contractsum_5085717L', \n        'credlmt_230A', 'credlmt_935A',\n        # 'debtoutstand_525A', 'debtoverdue_47A', 'dpdmax_139P', 'dpdmax_757P',\n    #    'instlamount_768A', 'instlamount_852A',\n    #    'interestrate_508L', 'monthlyinstlamount_332A',\n    #    'monthlyinstlamount_674A', \n            # good?\n       'nominalrate_281L', 'nominalrate_498L',\n       'numberofcontrsvalue_258L', 'numberofcontrsvalue_358L',\n       'numberofinstls_229L', 'numberofinstls_320L',\n       'numberofoutstandinstls_520L', 'numberofoutstandinstls_59L',\n       'numberofoverdueinstlmax_1039L', 'numberofoverdueinstlmax_1151L',\n       'numberofoverdueinstls_725L', 'numberofoverdueinstls_834L',\n            # bad?\n    #    'outstandingamount_354A', 'outstandingamount_362A', 'overdueamount_31A',\n    #    'overdueamount_659A', 'overdueamountmax2_14A', 'overdueamountmax2_398A',\n    #    'overdueamountmax_155A', 'overdueamountmax_35A',\n        # bad ?\n    #    'periodicityofpmts_1102L', 'periodicityofpmts_837L',\n    #    'prolongationcount_1120L', 'prolongationcount_599L',\n        # 520?\n    #    'residualamount_488A', 'residualamount_856A', 'totalamount_6A',\n    #    'totalamount_996A', 'totaldebtoverduevalue_178A',\n    #    'totaldebtoverduevalue_718A', 'totaloutstanddebtvalue_39A',\n    #    'totaloutstanddebtvalue_668A',\n       ]\n\n        # .697\n        # expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2] + [pl.std(col).alias(f\"std_{col}\") for col in cols2]\n        \n        # .696\n        # expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2]\n\n        # .697\n        # expr_3 = [pl.std(col).alias(f\"std_{col}\") for col in cols2]\n        \n        # .6985\n        # expr_3 = [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] + [pl.median(col).alias(f\"median_{col}\") for col in cols2]\n\n        # .696\n        # expr_3 = [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] \n\n        # .6981\n        # expr_3 = [pl.median(col).alias(f\"median_{col}\") for col in cols2]\n\n        # .696\n        # expr_3 = [pl.first(col).alias(f\"first_{col}\") for col in cols2] + [pl.last(col).alias(f\"last_{col}\") for col in cols2] # + \\\n        \n        # .696\n        # expr_3 = [pl.std(col).alias(f\"std_{col}\") for col in cols2] + [pl.median(col).alias(f\"median_{col}\") for col in cols2]\n\n        # .699\n        # expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2] + [pl.std(col).alias(f\"std_{col}\") for col in cols2] + \\\n        #     [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] + [pl.median(col).alias(f\"median_{col}\") for col in cols2]\n\n        expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2] + [pl.std(col).alias(f\"std_{col}\") for col in cols2] + \\\n            [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] + [pl.median(col).alias(f\"median_{col}\") for col in cols2] + \\\n            [pl.first(col).alias(f\"first_{col}\") for col in cols2] # + [pl.last(col).alias(f\"last_{col}\") for col in cols2] # not applied\n        \n        \n\n        # expr_3 = [pl.col(col).fill_null(strategy=\"zero\").apply(lambda x: x.max() - x.min()).alias(f\"max-min_gap_depth2_{col}\") for col in cols2]\n        return expr_1 + expr_2 + expr_3    \n\n\n    @staticmethod\n    def bureau_b1(df):  # 0.95에서 미적용 중 # 36500\n        # cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        # expr_1 = [pl.max(col).alias(f\"bureau_b1_max_{col}\") for col in cols]\n        # expr_2 = [pl.min(col).alias(f\"bureau_b1_min_{col}\") for col in cols]\n\n        # return expr_1 + expr_2 #  + expr_3\n        return []\n    \n    \n    @staticmethod\n    def bureau_b2(df):  # 0.95에서 미적용 중 # 36500\n        # cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        # expr_1 = [pl.max(col).alias(f\"bureau_b2_max_{col}\") for col in cols]\n        # expr_2 = [pl.min(col).alias(f\"bureau_b2_min_{col}\") for col in cols]\n\n        # return expr_1 + expr_2 #  + expr_3\n        return []\n\n\n    @staticmethod\n    def deposit_exprs(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] # + \\\n            # [pl.last(col).alias(f\"last_{col}\") for col in cols]\n            # [pl.mean(col).alias(f\"mean_{col}\") for col in cols] # + \\\n            # [pl.std(col).alias(f\"std_{col}\") for col in cols]  + \\\n             \n            # [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        # expr_2 = [pl.first('openingdate_857D').alias(f'first_openingdate_857D')] + [pl.last('openingdate_857D').alias(f'last_openingdate_857D')]\n        \n        return expr_1 # + expr_2 #+ expr_ngmax\n\n    @staticmethod\n    def debitcard_exprs(df):\n        # cols = [col for col in df.columns if (col[-1] in [\"A\"])]\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] \n            # [pl.mean(col).alias(f\"mean_{col}\") for col in cols] + \\\n            # [pl.std(col).alias(f\"std_{col}\") for col in cols]\n        # expr_2 = [pl.first('openingdate_857D').alias(f'first_openingdate_857D')] + [pl.last('openingdate_857D').alias(f'last_openingdate_857D')]\n        \n        return expr_1 # + expr_2 #+ expr_ngmax\n        # return expr_1\n\n\n    @staticmethod\n    def person_expr(df):\n        cols1 = ['empl_employedtotal_800L', 'empl_employedfrom_271D', 'empl_industry_691L', \n                 'familystate_447L', 'incometype_1044T', 'sex_738L', 'housetype_905L', 'housingtype_772L',\n                 'isreference_387L', 'birth_259D', ]\n        # cols1 = [col for col in df.columns]\n        expr_1 = [pl.first(col).alias(f\"first_{col}\") for col in cols1]\n        \n        expr_2 = [pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"), \n                  pl.col(\"mainoccupationinc_384A\").filter(pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")]\n        \n        # No Effect ...\n        # cols = ['personindex_1023L', 'persontype_1072L', 'persontype_792L']\n        # expr_3 = [pl.col(col).last().alias(f\"last_{col}\") for col in cols] + [pl.col(col).drop_nulls().mean().alias(f\"mean_{col}\") for col in cols]\n\n        # cols2 = [col for col in df.columns if col not in cols1]\n        # expr_4 = [pl.max(col).alias(f\"max_{col}\") for col in cols2] + [pl.min(col).alias(f\"min_{col}\") for col in cols2] #  good at cv, bad at lb ?\n            # [pl.col(col).drop_nulls().last().alias(f\"last_{col}\") for col in cols2] + [pl.col(col).drop_nulls().first().alias(f\"first_{col}\") for col in cols2] # no effect\n\n        return expr_1 + expr_2 # + expr_4 # + expr_3\n    \n    @staticmethod\n    def person_2_expr(df):\n        # cols = [col for col in df.columns]\n        cols = ['empls_economicalst_849M', 'empls_employedfrom_796D', 'empls_employer_name_740M'] # + \\\n            # ['relatedpersons_role_762T', 'conts_role_79M']\n            # ['addres_district_368M', 'addres_role_871L', 'addres_zip_823M']\n\n        expr_1 = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_2 = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n\n        # BAD\n        # expr_ngc = [pl.count(\"num_group2\").alias(f\"count_num_group2\")]\n        # cols2 = [col for col in df.columns if (col in (\"num_group1\", \"num_group2\"))]\n        # expr_ngmax = [pl.min(col).alias(f\"min_{col}\") for col in cols2] + [pl.max(col).alias(f\"max_{col}\") for col in cols2]\n\n        # cols2 = [col for col in df.columns if col not in cols]\n        # # expr_3 = [pl.max(col).alias(f\"max_{col}\") for col in cols2] + [pl.min(col).alias(f\"min_{col}\") for col in cols2] # no effect\n        # expr_3 = [pl.col(col).drop_nulls().last().alias(f\"last_{col}\") for col in cols2] # no effect\n\n        return expr_1 + expr_2 # + expr_3# + expr_ngc \n\n    @staticmethod\n    def other_expr(df):\n        expr_1 = [pl.first(col).alias(f\"__other_{col}\") for col in df.columns if ('num_group' not in col) and (col != 'case_id')]\n        # cols1 = ['amtdepositbalance_4809441A', 'amtdepositincoming_4809444A', 'amtdepositoutgoing_4809442A']\n        # expr_1 = [pl.last(col).alias(f\"last_{col}\") for col in cols1]\n        # cols2 = ['amtdebitincoming_4809443A', 'amtdebitoutgoing_4809440A']\n        # expr_3 = [(pl.col('amtdebitincoming_4809443A') - pl.col('amtdebitoutgoing_4809440A')).alias('amtdebit_incoming-outgoing')]\n        return expr_1 # + expr_2 + expr_3\n    \n    \n    @staticmethod\n    def tax_a_exprs(df):\n    # 选择列名符合条件的列\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        # 创建一组表达式，计算最大值、最小值、最后一个值、第一个值、平均值和标准差\n        expr_1 = [\n            pl.max(col).alias(f\"max_{col}\") for col in cols\n        ] + [\n            pl.min(col).alias(f\"min_{col}\") for col in cols\n        ] + [\n            pl.last(col).alias(f\"last_{col}\") for col in cols\n        ] + [\n            pl.first(col).alias(f\"first_{col}\") for col in cols\n        ] + [\n            pl.mean(col).alias(f\"mean_{col}\") for col in cols\n        ] + [\n            pl.std(col).alias(f\"std_{col}\") for col in cols\n        ]\n\n        # 计算特定列的最大值与最小值之差\n        expr_4 = [\n            (pl.col('amount_4527230A').fill_null(0).max() - pl.col('amount_4527230A').fill_null(0).min()).alias('max-min_gap_depth2_amount_4527230A')\n        ]\n\n        return expr_1 + expr_4\n\n\n\n    @staticmethod\n    def bureau_a2(df):\n    # 选择符合条件的列\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        # 创建表达式列表以计算最大值、最小值、平均值和标准差\n        expr_1 = [pl.max(col).alias(f\"max_depth2_{col}\") for col in cols]\n        expr_2 = [pl.min(col).alias(f\"min_depth2_{col}\") for col in cols]\n        expr_3 = [pl.mean(col).alias(f\"mean_depth2_{col}\") for col in cols] + \\\n                 [pl.std(col).alias(f\"std_{col}\") for col in cols]\n\n        # 计算特定列的最大值与最小值之差\n        expr_4 = [\n            (pl.max(col) - pl.min(col)).alias(f\"max-min_gap_depth2_{col}\")\n            for col in ['collater_valueofguarantee_1124L', 'pmts_dpd_1073P', 'pmts_overdue_1140A']\n        ]\n\n        # 计算特定组的计数\n        expr_ngc = [pl.count(\"num_group2\").alias(f\"count_depth2_a2_num_group2\")]\n\n        # 将所有表达式组合在一起\n        return expr_1 + expr_2 + expr_3 + expr_4 + expr_ngc\n\n    \n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df)\n\n        return exprs\n    \n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df)\n\n        return exprs\ndef agg_by_case(path, df):\n    path = str(path)\n    if '_applprev_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n\n#     elif '_applprev_2' in path:\n#         df = df.group_by(\"case_id\").agg(Aggregator.applprev2_exprs(df))\n\n    elif '_credit_bureau_a_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.bureau_a1(df))\n\n    elif '_credit_bureau_b_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.bureau_b1(df))\n\n    elif '_deposit_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.deposit_exprs(df))\n    elif '_debitcard_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.debitcard_exprs(df))\n        \n    elif '_tax_registry_a' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.tax_a_exprs(df))\n    elif '_tax_registry_b' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    elif '_tax_registry_c' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n    elif '_other_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.other_expr(df))\n    elif '_person_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.person_expr(df))\n    elif '_person_2' in path:\n        df = df.group_by(\"case_id\").agg(Aggregator.person_2_expr(df))\n\n    elif '_credit_bureau_a_2' in path:\n        df = df.group_by(\"case_id\").agg(Aggregator.bureau_a2(df))\n    elif '_credit_bureau_b_2' in path:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_file(path, depth=None): \n    df = pl.scan_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = agg_by_case(path, df)\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    print(regex_path)\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.scan_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = agg_by_case(path, df)\n        chunks.append(df)\n        #     del df\n        #     gc.collect()\n        #     print('delete chunk')\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base.with_columns(\n            decision_month = pl.col(\"date_decision\").dt.month(),\n            decision_weekday = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.collect().to_pandas()\n    print(df_data.info())\n\n    print(df_data.info())\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n        # cat_cols = [c for c in cat_cols if 'diff_' not in c]\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols\n\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n# ROOT            = Path(\"F:/data\")\n\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"\n# 加载模型\ncat_cols = joblib.load('/kaggle/input/datasets1/lgb_cat_cols.pickle')\n# ls_models = glob(os.path.join('/kaggle/input/lgb-fold', \"*.pkl\"))\n# models = [joblib.load(fn) for fn in ls_models]\nls_models = glob(os.path.join('/kaggle/input/lgbm/pytorch/default/1/', \"*.txt\"))\nmodels = [lgb.Booster(model_file=fn) for fn in ls_models]\nlen(cat_cols)\n# 测试数据\ndata_store_test = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2),\n    ]\n}\n# 处理数据集\ndf_test = feature_eng(**data_store_test)  # Load and process test data\n# print(df_test.shape)\n# print(\"---------------\")\ndf_test, _ = to_pandas(df_test)\n# print(df_test.shape)\n# print(\"tttttttttttttttt\")\ndf_test = reduce_mem_usage(df_test)\n# df_test.shape\n# print(\"kkkkkkkkkkkkkkkkkkkkkkk\")\ndel data_store_test\ngc.collect()\n# 存成excel，方便分批处理数据\ndf_test.drop(columns=[\"case_id\"]).to_csv('./test_data.csv', index=False)\ndel df_test\ngc.collect()\ndef set_categoricals(df_data, cat_cols):\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data\ndef lgb_prediction(feats, models):\n    predictions = np.zeros(len(feats))\n    for model in models:\n        p = model.predict(feats)\n        predictions += p/len(models)\n    return predictions\nCHUNK_SIZE = 10 ** 4\nreader = pd.read_csv('./test_data.csv', chunksize=CHUNK_SIZE)\ny_pred = []\nfor df_chunk in reader:\n    df_chunk = set_categoricals(df_chunk, cat_cols)\n    p = lgb_prediction(df_chunk, models)\n    y_pred.append(p)\ny_pred_lg = np.concatenate(y_pred, axis=0)\ndel reader\ngc.collect()\nprint(len(y_pred_lg))\nprint(\"-------------------------lg------------------------------------\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-30T12:53:40.614553Z","iopub.execute_input":"2024-10-30T12:53:40.615022Z","iopub.status.idle":"2024-10-30T12:53:46.028953Z","shell.execute_reply.started":"2024-10-30T12:53:40.614978Z","shell.execute_reply":"2024-10-30T12:53:46.027728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df1 = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\").set_index(\"case_id\")\nsubm_df1[\"score\"] = y_pred_lg","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:46.031801Z","iopub.execute_input":"2024-10-30T12:53:46.032440Z","iopub.status.idle":"2024-10-30T12:53:46.046843Z","shell.execute_reply.started":"2024-10-30T12:53:46.032352Z","shell.execute_reply":"2024-10-30T12:53:46.045000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(subm_df1).sort_index().to_csv('y_pred_lg.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:46.048817Z","iopub.execute_input":"2024-10-30T12:53:46.049286Z","iopub.status.idle":"2024-10-30T12:53:46.067821Z","shell.execute_reply.started":"2024-10-30T12:53:46.049234Z","shell.execute_reply":"2024-10-30T12:53:46.065882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y_pred_lg,subm_df1\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:46.069755Z","iopub.execute_input":"2024-10-30T12:53:46.071535Z","iopub.status.idle":"2024-10-30T12:53:46.262089Z","shell.execute_reply.started":"2024-10-30T12:53:46.071465Z","shell.execute_reply":"2024-10-30T12:53:46.260317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# from pathlib import Path\n# from datetime import datetime\n# import numpy as np\n# import pandas as pd\n# import polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# import joblib\nimport lightgbm as lgb\nimport torch\nimport torch.nn as nn\n\nfrom sklearn.model_selection import StratifiedGroupKFold\n# from sklearn.metrics import roc_auc_score\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\n# import warnings\n# warnings.simplefilter(action='ignore', category=FutureWarning)\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:19.735663Z\",\"iopub.execute_input\":\"2024-05-05T00:46:19.736281Z\",\"iopub.status.idle\":\"2024-05-05T00:46:20.928926Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:19.736254Z\",\"shell.execute_reply\":\"2024-05-05T00:46:20.928019Z\"}}\nfrom catboost import CatBoostClassifier, Pool\n\nsave_full_path = '/kaggle/input/cat-and-xgb/pytorch/default/1/catboost_model_fold_000.cbm'\n\n# load models\ncat_cols = joblib.load('/kaggle/input/datasets1/cat_cols.pickle')\n# ls_models = glob(os.path.join(f'{save_full_path}/', \"catboost_model_fold_*\"))\nmodels = [CatBoostClassifier().load_model(save_full_path)]\n\ncat_features = models[0].feature_names_\n# len(cat_features), len(models)\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:20.930161Z\",\"iopub.execute_input\":\"2024-05-05T00:46:20.930517Z\",\"iopub.status.idle\":\"2024-05-05T00:46:20.944098Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:20.930491Z\",\"shell.execute_reply\":\"2024-05-05T00:46:20.943206Z\"}}\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\n# %% [markdown] {\"papermill\":{\"duration\":0.006776,\"end_time\":\"2024-02-10T04:58:34.934015\",\"exception\":false,\"start_time\":\"2024-02-10T04:58:34.927239\",\"status\":\"completed\"},\"tags\":[]}\n# # Data collection\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:20.946731Z\",\"iopub.execute_input\":\"2024-05-05T00:46:20.947315Z\",\"iopub.status.idle\":\"2024-05-05T00:46:20.973026Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:20.947282Z\",\"shell.execute_reply\":\"2024-05-05T00:46:20.972161Z\"}}\nclass Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                # # TODO: Revisar el sentido de este filtro         \n                # if col[-1]=='M':\n                #     specific_value_ratio = df.filter(pl.col(col) == \"a55475b1\").height / df.height\n                #     if specific_value_ratio > 0.95:\n                #         df = df.drop(col)\n                        \n                if isnull > 0.95:\n                    df = df.drop(col)\n                    \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 50):\n                    df = df.drop(col)\n            \n            # # eliminate yaer, month feature\n            # if (col[-1] not in [\"P\", \"A\", \"L\", \"M\"]) and (('month_' in col) or ('year_' in col)):\n            #     df = df.drop(col)\n        \n        # for col in cols_drop:\n        #     if col in df.columns:\n        #         df = df.drop(col)\n            \n        return df\n\n    \n    # Añadidos los 3 siguientes metodos\n    @staticmethod\n    def reduce_memory_usage_pl(df):\n        \"\"\" Reduce memory usage by polars dataframe {df} with name {name} by changing its data types.\n            Original pandas version of this function: https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 \n        \"\"\"\n        print(f\"Memory usage of dataframe is {round(df.estimated_size('mb'), 2)} MB\")\n        \n        Numeric_Int_types = [pl.Int8, pl.Int16, pl.Int32, pl.Int64]\n        Numeric_Float_types = [pl.Float32, pl.Float64]    \n        \n        for col in df.columns:\n            if col == 'case_id': \n                continue\n            try:\n                col_type = df[col].dtype\n                \n                if col_type == pl.Categorical:\n                    continue\n                    \n                c_min = df[col].min()\n                c_max = df[col].max()\n                \n                if col_type in Numeric_Int_types:\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df = df.with_columns(df[col].cast(pl.Int8))\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df = df.with_columns(df[col].cast(pl.Int16))\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df = df.with_columns(df[col].cast(pl.Int32))\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df = df.with_columns(df[col].cast(pl.Int64))\n                \n                elif col_type in Numeric_Float_types:\n                    if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df = df.with_columns(df[col].cast(pl.Float32))\n                    else:\n                        pass\n                # elif col_type == pl.Utf8:\n                #     df = df.with_columns(df[col].cast(pl.Categorical))\n                else:\n                    pass\n            except:\n                pass\n        print(f\"Memory usage of dataframe became {round(df.estimated_size('mb'), 2)} MB\")\n        return df\n        \n    @staticmethod\n    def fill_missing_values(df):\n        num_cnt = 0\n        cat_cnt = 0\n        for col in df.columns:\n            if df[col].dtype.is_numeric():\n                df = df.with_columns(pl.col(col).fill_null(-1).alias(col))\n                num_cnt += 1\n            else:\n                df = df.with_columns(pl.col(col).fill_null(\"Missing\").alias(col))\n                cat_cnt += 1\n        print(\"num_cnt : \", num_cnt)\n        print(\"cat_cnt : \", cat_cnt)\n        return df\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:20.974176Z\",\"iopub.execute_input\":\"2024-05-05T00:46:20.974509Z\",\"iopub.status.idle\":\"2024-05-05T00:46:20.990447Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:20.974478Z\",\"shell.execute_reply\":\"2024-05-05T00:46:20.989676Z\"}}\nclass Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] # + [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        # expr_diff = [pl.col(col).diff().alias(f\"diff_{col}\") for col in cols]\n\n        return expr_max + [pl.mean(col).alias(f\"mean_{col}\") for col in cols] + [pl.std(col).alias(f\"std_{col}\") for col in cols] # + expr_diff\n    \n        # [pl.col(col).drop_nulls().last().alias(f\"last_{col}\") for col in cols]\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)] \n            \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] # + [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n\n        return expr_max # + [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.last(col).alias(f\"last_{col}\") for col in cols] + \\\n            [pl.n_unique(col).alias(f\"n_unique_{col}\") for col in cols] + \\\n            [pl.first(col).alias(f\"first_{col}\") for col in cols]  # High Value\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] + [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n\n        return expr_max # + [pl.mean(col).alias(f\"mean_{col}\") for col in cols] + [pl.std(col).alias(f\"std_{col}\") for col in cols]\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] # + [pl.n_unique(col).alias(f\"n_unique_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:20.991497Z\",\"iopub.execute_input\":\"2024-05-05T00:46:20.991771Z\",\"iopub.status.idle\":\"2024-05-05T00:46:21.007096Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:20.991749Z\",\"shell.execute_reply\":\"2024-05-05T00:46:21.006226Z\"}}\ndef read_file(path, depth=None):\n    df = pl.scan_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1]:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    elif depth in [2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n\n    \n    # Ensure df is a DataFrame\n    df = ensure_dataframe(df)\n    df = df.pipe(Pipeline.reduce_memory_usage_pl)\n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.scan_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1]:\n            df = df.collect().sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        elif depth in [2]:\n            df = df.collect().group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    # Ensure df is a DataFrame\n    df = ensure_dataframe(df)\n    df = df.pipe(Pipeline.reduce_memory_usage_pl)\n    \n    return df\n\ndef ensure_dataframe(df):\n    if isinstance(df, pl.LazyFrame):\n        return df.collect()\n    return df\n\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2, is_train=True):\n    df_base = (\n        df_base\n        .with_columns(\n            decision_month = pl.col(\"date_decision\").dt.month(),\n            decision_weekday = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    if is_train:\n        df_base = df_base.pipe(Pipeline.filter_cols)\n    df_base = df_base.pipe(Pipeline.fill_missing_values)\n    \n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    \n    df_data = df_data.to_pandas()\n#     df_data = df_data.collect().to_pandas()\n    print(df_data.info())\n\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols\n\n\n\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:21.008131Z\",\"iopub.execute_input\":\"2024-05-05T00:46:21.008411Z\",\"iopub.status.idle\":\"2024-05-05T00:46:21.020794Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:21.008389Z\",\"shell.execute_reply\":\"2024-05-05T00:46:21.019975Z\"}}\nfrom pathlib import Path\nfrom glob import glob\n\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:21.021879Z\",\"iopub.execute_input\":\"2024-05-05T00:46:21.022207Z\",\"iopub.status.idle\":\"2024-05-05T00:46:21.553731Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:21.022158Z\",\"shell.execute_reply\":\"2024-05-05T00:46:21.552792Z\"}}\ndata_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:21.554798Z\",\"iopub.execute_input\":\"2024-05-05T00:46:21.555088Z\",\"iopub.status.idle\":\"2024-05-05T00:46:22.189875Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:21.555063Z\",\"shell.execute_reply\":\"2024-05-05T00:46:22.189053Z\"}}\ndf_test = feature_eng(**data_store, is_train=False)\nprint(\"test data shape:\\t\", df_test.shape)\ndel data_store\ngc.collect()\n\n# %% [code] {\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:22.192593Z\",\"iopub.execute_input\":\"2024-05-05T00:46:22.192868Z\",\"iopub.status.idle\":\"2024-05-05T00:46:22.744567Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:22.192845Z\",\"shell.execute_reply\":\"2024-05-05T00:46:22.743676Z\"}}\n# df_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"vvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvv\")\ndf_test = df_test.select(['case_id', 'WEEK_NUM'] + cat_features)\nprint(\"----------------------------------------\")\n# print(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)\n\ndf_test, cat_cols = to_pandas(df_test,cat_cols)\n# df_test = reduce_mem_usage(df_test)\n\ngc.collect()\n\n# %% [markdown] {\"papermill\":{\"duration\":0.031773,\"end_time\":\"2024-02-10T05:23:03.16254\",\"exception\":false,\"start_time\":\"2024-02-10T05:23:03.130767\",\"status\":\"completed\"},\"tags\":[]}\n# # Prediction\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:22.745798Z\",\"iopub.execute_input\":\"2024-05-05T00:46:22.746081Z\",\"iopub.status.idle\":\"2024-05-05T00:46:22.751998Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:22.746057Z\",\"shell.execute_reply\":\"2024-05-05T00:46:22.751168Z\"}}\ndef cat_prediction(feats, models):\n    predictions = np.zeros(len(feats))\n    for model in models:\n        p = model.predict_proba(feats)[:, 1]\n        predictions += p/len(models)\n    return predictions\n\n# %% [code] {\"papermill\":{\"duration\":0.041537,\"end_time\":\"2024-02-10T05:23:03.235674\",\"exception\":false,\"start_time\":\"2024-02-10T05:23:03.194137\",\"status\":\"completed\"},\"tags\":[],\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:22.753123Z\",\"iopub.execute_input\":\"2024-05-05T00:46:22.753499Z\",\"iopub.status.idle\":\"2024-05-05T00:46:22.762590Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:22.753468Z\",\"shell.execute_reply\":\"2024-05-05T00:46:22.761734Z\"}}\ndef predict_proba_in_batches(models, data, batch_size=50000): # about 35min per 10k\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        \n        X_batch = data.iloc[start_idx:end_idx]\n        \n        # batch_probs = joblib.load(f'{automl_path}denselight_model.pkl').predict(X_batch).data.squeeze()\n        batch_probs = cat_prediction(X_batch, models)\n        \n        probabilities[start_idx:end_idx] = batch_probs\n        \n        del X_batch\n        gc.collect()\n\n    return probabilities\n\n# %% [code] {\"papermill\":{\"duration\":0.706127,\"end_time\":\"2024-02-10T05:23:03.973644\",\"exception\":false,\"start_time\":\"2024-02-10T05:23:03.267517\",\"status\":\"completed\"},\"tags\":[],\"jupyter\":{\"outputs_hidden\":false},\"execution\":{\"iopub.status.busy\":\"2024-05-05T00:46:22.763684Z\",\"iopub.execute_input\":\"2024-05-05T00:46:22.763949Z\",\"iopub.status.idle\":\"2024-05-05T00:46:23.107455Z\",\"shell.execute_reply.started\":\"2024-05-05T00:46:22.763926Z\",\"shell.execute_reply\":\"2024-05-05T00:46:23.106464Z\"}}\nX_test = df_test.drop(columns=[\"WEEK_NUM\"])\n\n\n\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", df_test.shape)\n\ny_pred_cat = pd.Series(predict_proba_in_batches(models, X_test), index=X_test.index)\ndel X_test,df_test\ngc.collect()\nprint(len(y_pred_cat))\nprint(\"-------------------------cat------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:46.266516Z","iopub.execute_input":"2024-10-30T12:53:46.267765Z","iopub.status.idle":"2024-10-30T12:53:54.435145Z","shell.execute_reply.started":"2024-10-30T12:53:46.267713Z","shell.execute_reply":"2024-10-30T12:53:54.433756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(y_pred_cat, columns=['predicted_values']).sort_index().to_csv('y_pred_cat.csv', index=True)\ndel y_pred_cat\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:54.436878Z","iopub.execute_input":"2024-10-30T12:53:54.437669Z","iopub.status.idle":"2024-10-30T12:53:54.656635Z","shell.execute_reply.started":"2024-10-30T12:53:54.437598Z","shell.execute_reply":"2024-10-30T12:53:54.655453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# import os\n# import gc\n# from glob import glob\n# from pathlib import Path\n# from datetime import datetime\n# import numpy as np\n# import pandas as pd\n# import polars as pl\n# import matplotlib.pyplot as plt\n# import seaborn as sns\n# import joblib\nimport lightgbm as lgb\n# import torch\n# import torch.nn as nn\n\n# from sklearn.model_selection import StratifiedGroupKFold\n# from sklearn.metrics import roc_auc_score\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\n# import warnings\n# warnings.simplefilter(action='ignore', category=FutureWarning)\n\n# Import XGBoost library\nimport xgboost as xgb\n\n# %% [code]\n\nsave_full_path = '/kaggle/input/cat-and-xgb/pytorch/default/1/xgboost_model_fold_1.model'\n\n# Load XGBoost model\nmodel = xgb.Booster()\nmodel.load_model(save_full_path)\n\ncat_cols = joblib.load('/kaggle/input/datasets1/cat_cols.pickle')\n\ncat_features_path = '/kaggle/input/datasets1/train_features.pkl'\ncat_features = joblib.load(cat_features_path)\n\n# %% [code]\ndef reduce_mem_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\n# %% [markdown]\n# # Data collection\n\n# %% [code]\nclass Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n                    \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 50):\n                    df = df.drop(col)\n            \n        return df\n\n    @staticmethod\n    def reduce_memory_usage_pl(df):\n        print(f\"Memory usage of dataframe is {round(df.estimated_size('mb'), 2)} MB\")\n        \n        Numeric_Int_types = [pl.Int8, pl.Int16, pl.Int32, pl.Int64]\n        Numeric_Float_types = [pl.Float32, pl.Float64]    \n        \n        for col in df.columns:\n            if col == 'case_id': \n                continue\n            try:\n                col_type = df[col].dtype\n                \n                if col_type == pl.Categorical:\n                    continue\n                    \n                c_min = df[col].min()\n                c_max = df[col].max()\n                \n                if col_type in Numeric_Int_types:\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df = df.with_columns(df[col].cast(pl.Int8))\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df = df.with_columns(df[col].cast(pl.Int16))\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df = df.with_columns(df[col].cast(pl.Int32))\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df = df.with_columns(df[col].cast(pl.Int64))\n                \n                elif col_type in Numeric_Float_types:\n                    if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df = df.with_columns(df[col].cast(pl.Float32))\n                    else:\n                        pass\n                else:\n                    pass\n            except:\n                pass\n        print(f\"Memory usage of dataframe became {round(df.estimated_size('mb'), 2)} MB\")\n        return df\n        \n    @staticmethod\n    def fill_missing_values(df):\n        num_cnt = 0\n        cat_cnt = 0\n        for col in df.columns:\n            if df[col].dtype.is_numeric():\n                df = df.with_columns(pl.col(col).fill_null(-1).alias(col))\n                num_cnt += 1\n            else:\n                df = df.with_columns(pl.col(col).fill_null(\"Missing\").alias(col))\n                cat_cnt += 1\n        print(\"num_cnt : \", num_cnt)\n        print(\"cat_cnt : \", cat_cnt)\n        return df\n\n# %% [code]\nclass Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        return expr_max + [pl.mean(col).alias(f\"mean_{col}\") for col in cols] + [pl.std(col).alias(f\"std_{col}\") for col in cols]\n    \n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)] \n            \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.last(col).alias(f\"last_{col}\") for col in cols] + \\\n            [pl.n_unique(col).alias(f\"n_unique_{col}\") for col in cols] + \\\n            [pl.first(col).alias(f\"first_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] + [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs\n\n# %% [code]\ndef read_file(path, depth=None):\n    df = pl.scan_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1]:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    elif depth in [2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n\n    \n    # Ensure df is a DataFrame\n    df = ensure_dataframe(df)\n    df = df.pipe(Pipeline.reduce_memory_usage_pl)\n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.scan_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1]:\n            df = df.collect().sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        elif depth in [2]:\n            df = df.collect().group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    # Ensure df is a DataFrame\n    df = ensure_dataframe(df)\n    df = df.pipe(Pipeline.reduce_memory_usage_pl)\n    \n    return df\n\ndef ensure_dataframe(df):\n    if isinstance(df, pl.LazyFrame):\n        return df.collect()\n    return df\n\n\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2, is_train=True):\n    df_base = (\n        df_base\n        .with_columns(\n            decision_month = pl.col(\"date_decision\").dt.month(),\n            decision_weekday = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    if is_train:\n        df_base = df_base.pipe(Pipeline.filter_cols)\n    df_base = df_base.pipe(Pipeline.fill_missing_values)\n    \n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols\n\n# %% [code]\nfrom pathlib import Path\nfrom glob import glob\n\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"\n\n# %% [code]\ndata_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}\n\n# %% [code]\ndf_test = feature_eng(**data_store, is_train=False)\nprint(\"--------------------------\")\nprint(\"test data shape:\\t\", df_test.shape)\nprint(\"vvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvv\")\ndel data_store\ngc.collect()\n\n# %% [code]\ndf_test = df_test.select(['case_id', 'WEEK_NUM'] + cat_features)\nprint(\"test data shape:\\t\", df_test.shape)\n\ndf_test, cat_cols = to_pandas(df_test,cat_cols)\nprint(\"*****************************\")\ndf_test.shape\nprint(\"vvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvvv\")\ngc.collect()\n\n# %% [markdown]\n# # Prediction\n\n# %% [code]\ndef xgb_prediction(dtest, model):\n\n    return model.predict(dtest)\n\n# %% [code]\ndef predict_proba_in_batches(model, data, batch_size=50000):\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        \n        dtest_batch = xgb.DMatrix(data.iloc[start_idx:end_idx],enable_categorical=True)\n\n        batch_probs = xgb_prediction(dtest_batch, model)\n        \n        probabilities[start_idx:end_idx] = batch_probs\n        \n        del dtest_batch\n        gc.collect()\n\n    return probabilities\n\n# %% [code]\nX_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\nprint(\"------------------\")\nX_test.shape\n# for col in X_test.columns:\n#     if X_test[col].dtype.name == 'category':\n# # 转换为category类型，确保数据类型正确\n#         X_test[col] = X_test[col].astype('category')\ndtest = xgb.DMatrix(X_test,enable_categorical=True)\nprint(\"X_test shape: \", df_test.shape)\n\n\n\ny_pred_xgb = pd.Series(predict_proba_in_batches(model, X_test), index=X_test.index)\n\ndel X_test,dtest,df_test\ngc.collect()\n\nprint(len(y_pred_xgb))\nprint(\"-------------------------xgb------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:54.662569Z","iopub.execute_input":"2024-10-30T12:53:54.663106Z","iopub.status.idle":"2024-10-30T12:53:57.479621Z","shell.execute_reply.started":"2024-10-30T12:53:54.663049Z","shell.execute_reply":"2024-10-30T12:53:57.478062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(y_pred_xgb, columns=['predicted_values']).sort_index().to_csv('y_pred_xgb.csv', index=True)\ndel y_pred_xgb\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:57.481487Z","iopub.execute_input":"2024-10-30T12:53:57.481976Z","iopub.status.idle":"2024-10-30T12:53:57.705867Z","shell.execute_reply.started":"2024-10-30T12:53:57.481925Z","shell.execute_reply":"2024-10-30T12:53:57.704465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file1 = 'y_pred_lg.csv'\nfile2 = 'y_pred_cat.csv'\nfile3 = 'y_pred_xgb.csv'\n\n# 初始化一个空的 DataFrame 用于存储最终结果\nfinal_result = pd.DataFrame()\n\n# 定义每次读取的行数\nchunksize = 20000\n\n# 逐块读取并计算平均值\nfor chunk1, chunk2, chunk3 in zip(\n    pd.read_csv(file1, chunksize=chunksize, index_col=0),  # 假设第一列是case_id并设置为索引\n    pd.read_csv(file2, chunksize=chunksize, index_col=0),\n    pd.read_csv(file3, chunksize=chunksize, index_col=0)\n):\n    # 确保每个块都有相同的索引\n    if not (chunk1.index.equals(chunk2.index) and chunk1.index.equals(chunk3.index)):\n        raise ValueError(\"索引不匹配！\")\n\n    # 计算平均值，确保列名正确\n    avg_chunk = (chunk1['score'] *0.5+ chunk2['predicted_values'] *0.40+ chunk3['predicted_values']*0.10) \n    avg_chunk = avg_chunk.rename('score').to_frame()  # 转换为 DataFrame\n\n    # 将结果拼接到 final_result 中\n    final_result = pd.concat([final_result, avg_chunk], axis=0)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:57.707716Z","iopub.execute_input":"2024-10-30T12:53:57.708211Z","iopub.status.idle":"2024-10-30T12:53:57.734293Z","shell.execute_reply.started":"2024-10-30T12:53:57.708157Z","shell.execute_reply":"2024-10-30T12:53:57.732418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_result.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T12:53:57.736141Z","iopub.execute_input":"2024-10-30T12:53:57.736662Z","iopub.status.idle":"2024-10-30T12:53:57.745160Z","shell.execute_reply.started":"2024-10-30T12:53:57.736606Z","shell.execute_reply":"2024-10-30T12:53:57.743681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}