{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":166996856,"sourceType":"kernelVersion"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Created by <a href=\"https://github.com/yunsuxiaozi\">yunsuxiaozi</a> 2024/4/6","metadata":{}},{"cell_type":"code","source":"import polars as pl#和pandas类似,但是处理大型数据集有更好的性能.\n#necessary\nimport pandas as pd#导入csv文件的库\nimport numpy as np#进行矩阵运算的库\n#model lgb分类模型,日志评估,早停防止过拟合\nfrom  lightgbm import LGBMClassifier,log_evaluation,early_stopping\n#metric\nfrom sklearn.metrics import roc_auc_score#导入roc_auc曲线\n#KFold是直接分成k折,StratifiedKFold还要考虑每种类别的占比\nfrom sklearn.model_selection import StratifiedKFold\nimport dill#对对象进行序列化和反序列化(例如保存和加载树模型)\nimport gc#垃圾回收模块\nimport time#标准库的时间模块\n#为了方便后期调用训练的模型时不会调用错版本,提供模型训练的时间\n#time.strftime()函数用于将时间对象格式化为字符串，time.localtime()函数返回表示当前本地时间的time.struct_time对象\ncurrent_time = time.strftime(\"%Y-%m-%d %H:%M:%S\", time.localtime())\nprint(\"this notebook training time is \", current_time)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#config\nclass Config():\n    seed=2024\n    num_folds=10\n    TARGET_NAME ='target'\n    batch_size=1000#由于不知道测试数据的大小,所以分批次放入模型.\n    #非类别型字符串,nunique=1,isna.mean()>0.95的列\n    #日期列目前计算了和date_decision的gap_day,如果gap_day要drop的话,还不如直接drop原始日期列.\n    drop_cols=['max_applprev1_cancelreason_3545846M', 'last_applprev1_cancelreason_3545846M', 'max_applprev1_district_544M', 'last_applprev1_district_544M', 'max_applprev1_isbidproduct_390L', 'last_applprev1_isbidproduct_390L', 'max_applprev1_isdebitcard_527L', 'last_applprev1_isdebitcard_527L', 'max_applprev1_profession_152M', 'last_applprev1_profession_152M', 'last_applprev1_revolvingaccount_394A', 'last_applprev2_credacc_cards_status_52L', 'last_credit_bureau_a_1_annualeffectiverate_199L', 'last_credit_bureau_a_1_annualeffectiverate_63L', 'max_credit_bureau_a_1_classificationofcontr_400M', 'last_credit_bureau_a_1_classificationofcontr_400M', 'max_credit_bureau_a_1_contractst_964M', 'last_credit_bureau_a_1_contractst_964M', 'last_credit_bureau_a_1_contractsum_5085717L', 'last_credit_bureau_a_1_credlmt_230A', 'last_credit_bureau_a_1_credlmt_935A', 'last_credit_bureau_a_1_debtoutstand_525A', 'last_credit_bureau_a_1_debtoverdue_47A', 'last_credit_bureau_a_1_dpdmax_139P', 'last_credit_bureau_a_1_dpdmax_757P', 'last_credit_bureau_a_1_dpdmaxdatemonth_442T', 'last_credit_bureau_a_1_dpdmaxdatemonth_89T', 'last_credit_bureau_a_1_dpdmaxdateyear_596T', 'last_credit_bureau_a_1_dpdmaxdateyear_896T', 'max_credit_bureau_a_1_financialinstitution_382M', 'last_credit_bureau_a_1_financialinstitution_382M', 'max_credit_bureau_a_1_financialinstitution_591M', 'last_credit_bureau_a_1_instlamount_768A', 'last_credit_bureau_a_1_instlamount_852A', 'max_credit_bureau_a_1_interestrate_508L', 'last_credit_bureau_a_1_interestrate_508L', 'last_credit_bureau_a_1_monthlyinstlamount_332A', 'last_credit_bureau_a_1_monthlyinstlamount_674A', 'last_credit_bureau_a_1_nominalrate_281L', 'last_credit_bureau_a_1_nominalrate_498L', 'last_credit_bureau_a_1_numberofcontrsvalue_258L', 'last_credit_bureau_a_1_numberofcontrsvalue_358L', 'last_credit_bureau_a_1_numberofinstls_229L', 'last_credit_bureau_a_1_numberofinstls_320L', 'last_credit_bureau_a_1_numberofoutstandinstls_520L', 'last_credit_bureau_a_1_numberofoutstandinstls_59L', 'last_credit_bureau_a_1_numberofoverdueinstlmax_1039L', 'last_credit_bureau_a_1_numberofoverdueinstlmax_1151L', 'last_credit_bureau_a_1_numberofoverdueinstls_725L', 'last_credit_bureau_a_1_numberofoverdueinstls_834L', 'last_credit_bureau_a_1_outstandingamount_354A', 'last_credit_bureau_a_1_outstandingamount_362A', 'last_credit_bureau_a_1_overdueamount_31A', 'last_credit_bureau_a_1_overdueamount_659A', 'last_credit_bureau_a_1_overdueamountmax2_14A', 'last_credit_bureau_a_1_overdueamountmax2_398A', 'last_credit_bureau_a_1_overdueamountmax_155A', 'last_credit_bureau_a_1_overdueamountmax_35A', 'last_credit_bureau_a_1_overdueamountmaxdatemonth_284T', 'last_credit_bureau_a_1_overdueamountmaxdatemonth_365T', 'last_credit_bureau_a_1_overdueamountmaxdateyear_2T', 'last_credit_bureau_a_1_overdueamountmaxdateyear_994T', 'last_credit_bureau_a_1_periodicityofpmts_1102L', 'last_credit_bureau_a_1_periodicityofpmts_837L', 'last_credit_bureau_a_1_prolongationcount_1120L', 'max_credit_bureau_a_1_prolongationcount_599L', 'last_credit_bureau_a_1_prolongationcount_599L', 'last_credit_bureau_a_1_residualamount_488A', 'last_credit_bureau_a_1_residualamount_856A', 'last_credit_bureau_a_1_subjectrole_182M', 'last_credit_bureau_a_1_totalamount_6A', 'last_credit_bureau_a_1_totalamount_996A', 'last_credit_bureau_a_1_totaldebtoverduevalue_178A', 'last_credit_bureau_a_1_totaldebtoverduevalue_718A', 'last_credit_bureau_a_1_totaloutstanddebtvalue_39A', 'last_credit_bureau_a_1_totaloutstanddebtvalue_668A', 'max_collater_typofvalofguarant_298M', 'last_collater_typofvalofguarant_298M', 'std_collater_typofvalofguarant_298M', 'max_pmts_month_158T', 'last_pmts_month_158T', 'std_pmts_month_158T', 'max_pmts_month_706T', 'last_pmts_month_706T', 'std_pmts_month_706T', 'max_bureau_b_1_amount_1115A', 'last_bureau_b_1_amount_1115A', 'max_bureau_b_1_classificationofcontr_1114M', 'last_bureau_b_1_classificationofcontr_1114M', 'max_bureau_b_1_contractst_516M', 'last_bureau_b_1_contractst_516M', 'max_bureau_b_1_contracttype_653M', 'last_bureau_b_1_contracttype_653M', 'max_bureau_b_1_credlmt_1052A', 'last_bureau_b_1_credlmt_1052A', 'max_bureau_b_1_credlmt_228A', 'last_bureau_b_1_credlmt_228A', 'max_bureau_b_1_credlmt_3940954A', 'last_bureau_b_1_credlmt_3940954A', 'max_bureau_b_1_credor_3940957M', 'last_bureau_b_1_credor_3940957M', 'max_bureau_b_1_credquantity_1099L', 'last_bureau_b_1_credquantity_1099L', 'max_bureau_b_1_credquantity_984L', 'last_bureau_b_1_credquantity_984L', 'max_bureau_b_1_debtpastduevalue_732A', 'last_bureau_b_1_debtpastduevalue_732A', 'max_bureau_b_1_debtvalue_227A', 'last_bureau_b_1_debtvalue_227A', 'max_bureau_b_1_dpd_550P', 'last_bureau_b_1_dpd_550P', 'max_bureau_b_1_dpd_733P', 'last_bureau_b_1_dpd_733P', 'max_bureau_b_1_dpdmax_851P', 'last_bureau_b_1_dpdmax_851P', 'max_bureau_b_1_dpdmaxdatemonth_804T', 'last_bureau_b_1_dpdmaxdatemonth_804T', 'max_bureau_b_1_dpdmaxdateyear_742T', 'last_bureau_b_1_dpdmaxdateyear_742T', 'max_bureau_b_1_installmentamount_644A', 'last_bureau_b_1_installmentamount_644A', 'max_bureau_b_1_installmentamount_833A', 'last_bureau_b_1_installmentamount_833A', 'max_bureau_b_1_instlamount_892A', 'last_bureau_b_1_instlamount_892A', 'max_bureau_b_1_interesteffectiverate_369L', 'last_bureau_b_1_interesteffectiverate_369L', 'max_bureau_b_1_interestrateyearly_538L', 'last_bureau_b_1_interestrateyearly_538L', 'max_bureau_b_1_maxdebtpduevalodued_3940955A', 'last_bureau_b_1_maxdebtpduevalodued_3940955A', 'max_bureau_b_1_num_group1', 'last_bureau_b_1_num_group1', 'max_bureau_b_1_numberofinstls_810L', 'last_bureau_b_1_numberofinstls_810L', 'max_bureau_b_1_overdueamountmax_950A', 'last_bureau_b_1_overdueamountmax_950A', 'max_bureau_b_1_overdueamountmaxdatemonth_494T', 'last_bureau_b_1_overdueamountmaxdatemonth_494T', 'max_bureau_b_1_overdueamountmaxdateyear_432T', 'last_bureau_b_1_overdueamountmaxdateyear_432T', 'max_bureau_b_1_periodicityofpmts_997L', 'last_bureau_b_1_periodicityofpmts_997L', 'max_bureau_b_1_periodicityofpmts_997M', 'last_bureau_b_1_periodicityofpmts_997M', 'max_bureau_b_1_pmtdaysoverdue_1135P', 'last_bureau_b_1_pmtdaysoverdue_1135P', 'max_bureau_b_1_pmtmethod_731M', 'last_bureau_b_1_pmtmethod_731M', 'max_bureau_b_1_pmtnumpending_403L', 'last_bureau_b_1_pmtnumpending_403L', 'max_bureau_b_1_purposeofcred_722M', 'last_bureau_b_1_purposeofcred_722M', 'max_bureau_b_1_residualamount_1093A', 'last_bureau_b_1_residualamount_1093A', 'max_bureau_b_1_residualamount_127A', 'last_bureau_b_1_residualamount_127A', 'max_bureau_b_1_residualamount_3940956A', 'last_bureau_b_1_residualamount_3940956A', 'max_bureau_b_1_subjectrole_326M', 'last_bureau_b_1_subjectrole_326M', 'max_bureau_b_1_subjectrole_43M', 'last_bureau_b_1_subjectrole_43M', 'max_bureau_b_1_totalamount_503A', 'last_bureau_b_1_totalamount_503A', 'max_bureau_b_1_totalamount_881A', 'last_bureau_b_1_totalamount_881A', 'max_bureau_b_2_num_group1', 'last_bureau_b_2_num_group1', 'max_bureau_b_2_num_group2', 'last_bureau_b_2_num_group2', 'max_bureau_b_2_pmts_dpdvalue_108P', 'last_bureau_b_2_pmts_dpdvalue_108P', 'max_bureau_b_2_pmts_pmtsoverdue_635A', 'last_bureau_b_2_pmts_pmtsoverdue_635A', 'max_debitcard_last180dayaveragebalance_704A', 'last_debitcard_last180dayaveragebalance_704A', 'max_debitcard_last180dayturnover_1134A', 'last_debitcard_last180dayturnover_1134A', 'max_debitcard_last30dayturnover_651A', 'last_debitcard_last30dayturnover_651A', 'max_other_amtdebitincoming_4809443A', 'last_other_amtdebitincoming_4809443A', 'max_other_amtdebitoutgoing_4809440A', 'last_other_amtdebitoutgoing_4809440A', 'max_other_amtdepositbalance_4809441A', 'last_other_amtdepositbalance_4809441A', 'max_other_amtdepositincoming_4809444A', 'last_other_amtdepositincoming_4809444A', 'max_other_amtdepositoutgoing_4809442A', 'last_other_amtdepositoutgoing_4809442A', 'max_other_num_group1', 'last_other_num_group1', 'max_person1_childnum_185L', 'last_person1_childnum_185L', 'max_person1_contaddr_district_15M', 'last_person1_contaddr_district_15M', 'max_person1_contaddr_matchlist_1032L', 'last_person1_contaddr_matchlist_1032L', 'last_person1_contaddr_smempladdr_334L', 'max_person1_contaddr_zipcode_807M', 'last_person1_contaddr_zipcode_807M', 'last_person1_empl_employedtotal_800L', 'last_person1_empl_industry_691L', 'last_person1_empladdr_district_926M', 'last_person1_empladdr_zipcode_114M', 'last_person1_familystate_447L', 'max_person1_gender_992L', 'last_person1_gender_992L', 'last_person1_housetype_905L', 'max_person1_housingtype_772L', 'last_person1_housingtype_772L', 'max_person1_isreference_387L', 'last_person1_isreference_387L', 'max_person1_maritalst_703L', 'last_person1_maritalst_703L', 'max_person1_registaddr_district_1083M', 'last_person1_registaddr_district_1083M', 'max_person1_registaddr_zipcode_184M', 'last_person1_registaddr_zipcode_184M', 'max_person1_remitter_829L', 'last_person1_remitter_829L', 'max_person1_role_993L', 'last_person1_role_993L', 'last_person1_safeguarantyflag_411L', 'last_person1_sex_738L', 'max_person2_addres_district_368M', 'last_person2_addres_district_368M', 'max_person2_addres_role_871L', 'last_person2_addres_role_871L', 'max_person2_addres_zip_823M', 'last_person2_addres_zip_823M', 'max_person2_empls_employer_name_740M', 'last_person2_empls_employer_name_740M', 'max_person2_relatedpersons_role_762T', 'last_person2_relatedpersons_role_762T', 'amtinstpaidbefduel24m_4187115A', 'avgdbddpdlast3m_4187120P', 'avgdbdtollast24m_4525197P', 'avglnamtstart24m_4525187A', 'avgoutstandbalancel6m_4187114A', 'avgpmtlast12m_4525200A', 'bankacctype_710L', 'cardtype_51L', 'clientscnt_136L', 'commnoinclast6m_3546845L', 'deferredmnthsnum_166L', 'equalitydataagreement_891L', 'equalityempfrom_62L', 'interestrategrace_34L', 'isbidproductrequest_292L', 'isdebitcard_729L', 'lastapprcommoditytypec_5251766M', 'lastcancelreason_561M', 'lastdependentsnum_448L', 'lastotherinc_902A', 'lastotherlnsexpense_631A', 'lastrejectcommodtypec_5251769M', 'mastercontrelectronic_519L', 'mastercontrexist_109L', 'maxannuity_4075009A', 'maxdbddpdtollast6m_4187119P', 'maxlnamtstart6m_4525199A', 'maxoutstandbalancel12m_4187113A', 'maxpmtlast3m_4525190A', 'mindbdtollast24m_4525191P', 'numinstlswithdpd5_4187116L', 'numinstmatpaidtearly2d_4499204L', 'numinstpaid_4499208L', 'numinstpaidearly3dest_4493216L', 'numinstpaidearly5dest_4493211L', 'numinstpaidearly5dobd_4499205L', 'numinstpaidearlyest_4493214L', 'numinstpaidlastcontr_4325080L', 'numinstregularpaidest_4493210L', 'numinsttopaygrest_4493213L', 'numinstunpaidmaxest_4493212L', 'opencred_647L', 'paytype1st_925L', 'paytype_783L', 'previouscontdistrict_112M', 'sumoutstandtotalest_4493215A', 'totinstallast1m_4525188A', 'typesuite_864L', 'max_static_cb_contractssum_5085716L', 'last_static_cb_contractssum_5085716L', 'max_static_cb_description_5085714M', 'last_static_cb_description_5085714M', 'max_static_cb_for3years_128L', 'last_static_cb_for3years_128L', 'max_static_cb_for3years_504L', 'last_static_cb_for3years_504L', 'max_static_cb_for3years_584L', 'last_static_cb_for3years_584L', 'max_static_cb_formonth_118L', 'last_static_cb_formonth_118L', 'max_static_cb_formonth_206L', 'last_static_cb_formonth_206L', 'max_static_cb_formonth_535L', 'last_static_cb_formonth_535L', 'max_static_cb_forquarter_1017L', 'last_static_cb_forquarter_1017L', 'max_static_cb_forquarter_462L', 'last_static_cb_forquarter_462L', 'max_static_cb_forquarter_634L', 'last_static_cb_forquarter_634L', 'max_static_cb_fortoday_1092L', 'last_static_cb_fortoday_1092L', 'max_static_cb_forweek_1077L', 'last_static_cb_forweek_1077L', 'max_static_cb_forweek_528L', 'last_static_cb_forweek_528L', 'max_static_cb_forweek_601L', 'last_static_cb_forweek_601L', 'max_static_cb_foryear_618L', 'last_static_cb_foryear_618L', 'max_static_cb_foryear_818L', 'last_static_cb_foryear_818L', 'max_static_cb_foryear_850L', 'last_static_cb_foryear_850L', 'max_static_cb_pmtaverage_4527227A', 'last_static_cb_pmtaverage_4527227A', 'max_static_cb_pmtaverage_4955615A', 'last_static_cb_pmtaverage_4955615A', 'max_static_cb_pmtcount_4955617L', 'last_static_cb_pmtcount_4955617L', 'max_static_cb_riskassesment_302T', 'last_static_cb_riskassesment_302T', 'max_static_cb_riskassesment_940T', 'last_static_cb_riskassesment_940T', 'max_tax_a_name_4527232M', 'last_tax_a_name_4527232M', 'max_tax_b_name_4917606M', 'last_tax_b_name_4917606M', 'max_tax_c_employername_160M', 'last_tax_c_employername_160M', 'last_credit_bureau_a_1_dateofcredend_289D', 'last_credit_bureau_a_1_dateofcredend_353D', 'last_credit_bureau_a_1_dateofcredstart_181D', 'last_credit_bureau_a_1_dateofcredstart_739D', 'last_credit_bureau_a_1_dateofrealrepmt_138D', 'last_credit_bureau_a_1_lastupdate_1112D', 'last_credit_bureau_a_1_lastupdate_388D', 'last_credit_bureau_a_1_numberofoverdueinstlmaxdat_148D', 'last_credit_bureau_a_1_numberofoverdueinstlmaxdat_641D', 'last_credit_bureau_a_1_overdueamountmax2date_1002D', 'last_credit_bureau_a_1_overdueamountmax2date_1142D', 'max_bureau_b_1_contractdate_551D', 'last_bureau_b_1_contractdate_551D', 'max_bureau_b_1_contractmaturitydate_151D', 'last_bureau_b_1_contractmaturitydate_151D', 'max_bureau_b_1_lastupdate_260D', 'last_bureau_b_1_lastupdate_260D', 'max_bureau_b_2_pmts_date_1107D', 'last_bureau_b_2_pmts_date_1107D', 'max_deposit_contractenddate_991D', 'last_deposit_contractenddate_991D', 'max_person1_birthdate_87D', 'last_person1_birthdate_87D', 'last_person1_empl_employedfrom_271D', 'max_person2_empls_employedfrom_796D', 'last_person2_empls_employedfrom_796D', 'lastrepayingdate_696D', 'payvacationpostpone_4187118D', 'max_static_cb_assignmentdate_4955616D', 'last_static_cb_assignmentdate_4955616D', 'max_static_cb_dateofbirth_342D', 'last_static_cb_dateofbirth_342D', 'std_applprev1_cancelreason_3545846M', 'std_applprev1_credacc_actualbalance_314A', 'std_applprev1_credacc_maxhisbal_375A', 'std_applprev1_credacc_minhisbal_90A', 'std_applprev1_credacc_status_367L', 'std_applprev1_credacc_transactions_402L', 'std_applprev1_credtype_587L', 'std_applprev1_district_544M', 'std_applprev1_education_1138M', 'std_applprev1_familystate_726L', 'std_applprev1_inittransactioncode_279L', 'std_applprev1_isbidproduct_390L', 'std_applprev1_isdebitcard_527L', 'std_applprev1_postype_4733339M', 'std_applprev1_profession_152M', 'std_applprev1_rejectreason_755M', 'std_applprev1_rejectreasonclient_4145042M', 'std_applprev1_revolvingaccount_394A', 'std_applprev1_status_219L', 'std_applprev2_cacccardblochreas_147M', 'std_applprev2_conts_type_509L', 'std_applprev2_credacc_cards_status_52L', 'std_credit_bureau_a_1_annualeffectiverate_63L', 'std_credit_bureau_a_1_classificationofcontr_13M', 'std_credit_bureau_a_1_classificationofcontr_400M', 'std_credit_bureau_a_1_contractst_545M', 'std_credit_bureau_a_1_contractst_964M', 'std_credit_bureau_a_1_debtoutstand_525A', 'std_credit_bureau_a_1_debtoverdue_47A', 'std_credit_bureau_a_1_description_351M', 'std_credit_bureau_a_1_financialinstitution_382M', 'std_credit_bureau_a_1_financialinstitution_591M', 'std_credit_bureau_a_1_interestrate_508L', 'std_credit_bureau_a_1_numberofcontrsvalue_258L', 'std_credit_bureau_a_1_numberofcontrsvalue_358L', 'std_credit_bureau_a_1_prolongationcount_1120L', 'std_credit_bureau_a_1_prolongationcount_599L', 'std_credit_bureau_a_1_purposeofcred_426M', 'std_credit_bureau_a_1_purposeofcred_874M', 'std_credit_bureau_a_1_subjectrole_182M', 'std_credit_bureau_a_1_subjectrole_93M', 'std_credit_bureau_a_1_totaldebtoverduevalue_178A', 'std_credit_bureau_a_1_totaldebtoverduevalue_718A', 'std_credit_bureau_a_1_totaloutstanddebtvalue_39A', 'std_credit_bureau_a_1_totaloutstanddebtvalue_668A', 'std_bureau_b_1_amount_1115A', 'std_bureau_b_1_classificationofcontr_1114M', 'std_bureau_b_1_contractst_516M', 'std_bureau_b_1_contracttype_653M', 'std_bureau_b_1_credlmt_1052A', 'std_bureau_b_1_credlmt_228A', 'std_bureau_b_1_credlmt_3940954A', 'std_bureau_b_1_credor_3940957M', 'std_bureau_b_1_credquantity_1099L', 'std_bureau_b_1_credquantity_984L', 'std_bureau_b_1_debtpastduevalue_732A', 'std_bureau_b_1_debtvalue_227A', 'std_bureau_b_1_dpd_550P', 'std_bureau_b_1_dpd_733P', 'std_bureau_b_1_dpdmax_851P', 'std_bureau_b_1_dpdmaxdatemonth_804T', 'std_bureau_b_1_dpdmaxdateyear_742T', 'std_bureau_b_1_installmentamount_644A', 'std_bureau_b_1_installmentamount_833A', 'std_bureau_b_1_instlamount_892A', 'std_bureau_b_1_interesteffectiverate_369L', 'std_bureau_b_1_interestrateyearly_538L', 'std_bureau_b_1_maxdebtpduevalodued_3940955A', 'std_bureau_b_1_num_group1', 'std_bureau_b_1_numberofinstls_810L', 'std_bureau_b_1_overdueamountmax_950A', 'std_bureau_b_1_overdueamountmaxdatemonth_494T', 'std_bureau_b_1_overdueamountmaxdateyear_432T', 'std_bureau_b_1_periodicityofpmts_997L', 'std_bureau_b_1_periodicityofpmts_997M', 'std_bureau_b_1_pmtdaysoverdue_1135P', 'std_bureau_b_1_pmtmethod_731M', 'std_bureau_b_1_pmtnumpending_403L', 'std_bureau_b_1_purposeofcred_722M', 'std_bureau_b_1_residualamount_1093A', 'std_bureau_b_1_residualamount_127A', 'std_bureau_b_1_residualamount_3940956A', 'std_bureau_b_1_subjectrole_326M', 'std_bureau_b_1_subjectrole_43M', 'std_bureau_b_1_totalamount_503A', 'std_bureau_b_1_totalamount_881A', 'std_bureau_b_2_num_group1', 'std_bureau_b_2_num_group2', 'std_bureau_b_2_pmts_dpdvalue_108P', 'std_bureau_b_2_pmts_pmtsoverdue_635A', 'std_debitcard_last180dayaveragebalance_704A', 'std_debitcard_last180dayturnover_1134A', 'std_debitcard_last30dayturnover_651A', 'std_debitcard_num_group1', 'std_deposit_amount_416A', 'std_deposit_num_group1', 'std_other_amtdebitincoming_4809443A', 'std_other_amtdebitoutgoing_4809440A', 'std_other_amtdepositbalance_4809441A', 'std_other_amtdepositincoming_4809444A', 'std_other_amtdepositoutgoing_4809442A', 'std_other_num_group1', 'std_person1_childnum_185L', 'std_person1_contaddr_district_15M', 'std_person1_contaddr_matchlist_1032L', 'std_person1_contaddr_smempladdr_334L', 'std_person1_contaddr_zipcode_807M', 'std_person1_education_927M', 'std_person1_empl_employedtotal_800L', 'std_person1_empl_industry_691L', 'std_person1_empladdr_district_926M', 'std_person1_empladdr_zipcode_114M', 'std_person1_familystate_447L', 'std_person1_gender_992L', 'std_person1_housetype_905L', 'std_person1_housingtype_772L', 'std_person1_incometype_1044T', 'std_person1_isreference_387L', 'std_person1_language1_981M', 'std_person1_mainoccupationinc_384A', 'std_person1_maritalst_703L', 'std_person1_registaddr_district_1083M', 'std_person1_registaddr_zipcode_184M', 'std_person1_relationshiptoclient_415T', 'std_person1_relationshiptoclient_642T', 'std_person1_remitter_829L', 'std_person1_role_1084L', 'std_person1_role_993L', 'std_person1_safeguarantyflag_411L', 'std_person1_sex_738L', 'std_person1_type_25L', 'std_person2_addres_district_368M', 'std_person2_addres_role_871L', 'std_person2_addres_zip_823M', 'std_person2_conts_role_79M', 'std_person2_empls_economicalst_849M', 'std_person2_empls_employer_name_740M', 'std_person2_relatedpersons_role_762T', 'std_static_cb_contractssum_5085716L', 'std_static_cb_days120_123L', 'std_static_cb_days180_256L', 'std_static_cb_days30_165L', 'std_static_cb_days360_512L', 'std_static_cb_days90_310L', 'std_static_cb_description_5085714M', 'std_static_cb_education_1103M', 'std_static_cb_education_88M', 'std_static_cb_firstquarter_103L', 'std_static_cb_for3years_128L', 'std_static_cb_for3years_504L', 'std_static_cb_for3years_584L', 'std_static_cb_formonth_118L', 'std_static_cb_formonth_206L', 'std_static_cb_formonth_535L', 'std_static_cb_forquarter_1017L', 'std_static_cb_forquarter_462L', 'std_static_cb_forquarter_634L', 'std_static_cb_fortoday_1092L', 'std_static_cb_forweek_1077L', 'std_static_cb_forweek_528L', 'std_static_cb_forweek_601L', 'std_static_cb_foryear_618L', 'std_static_cb_foryear_818L', 'std_static_cb_foryear_850L', 'std_static_cb_fourthquarter_440L', 'std_static_cb_maritalst_385M', 'std_static_cb_maritalst_893M', 'std_static_cb_numberofqueries_373L', 'std_static_cb_pmtaverage_3A', 'std_static_cb_pmtaverage_4527227A', 'std_static_cb_pmtaverage_4955615A', 'std_static_cb_pmtcount_4527229L', 'std_static_cb_pmtcount_4955617L', 'std_static_cb_pmtcount_693L', 'std_static_cb_pmtscount_423L', 'std_static_cb_pmtssum_45A', 'std_static_cb_requesttype_4525192L', 'std_static_cb_riskassesment_302T', 'std_static_cb_riskassesment_940T', 'std_static_cb_secondquarter_766L', 'std_static_cb_thirdquarter_1082L', 'std_tax_a_name_4527232M', 'std_tax_b_name_4917606M', 'std_tax_c_employername_160M', 'std_applprev1_approvaldate_319D', 'std_applprev1_creationdate_885D', 'std_applprev1_dateactivated_425D', 'std_applprev1_dtlastpmt_581D', 'std_applprev1_dtlastpmtallstes_3545839D', 'std_applprev1_employedfrom_700D', 'std_applprev1_firstnonzeroinstldate_307D', 'std_credit_bureau_a_1_dateofcredend_289D', 'std_credit_bureau_a_1_dateofcredend_353D', 'std_credit_bureau_a_1_dateofcredstart_181D', 'std_credit_bureau_a_1_dateofcredstart_739D', 'std_credit_bureau_a_1_dateofrealrepmt_138D', 'std_credit_bureau_a_1_lastupdate_1112D', 'std_credit_bureau_a_1_lastupdate_388D', 'std_credit_bureau_a_1_numberofoverdueinstlmaxdat_148D', 'std_credit_bureau_a_1_numberofoverdueinstlmaxdat_641D', 'std_credit_bureau_a_1_overdueamountmax2date_1002D', 'std_credit_bureau_a_1_overdueamountmax2date_1142D', 'std_credit_bureau_a_1_refreshdate_3813885D', 'std_bureau_b_1_contractdate_551D', 'std_bureau_b_1_contractmaturitydate_151D', 'std_bureau_b_1_lastupdate_260D', 'std_bureau_b_2_pmts_date_1107D', 'std_debitcard_openingdate_857D', 'std_deposit_contractenddate_991D', 'std_deposit_openingdate_313D', 'std_person1_birth_259D', 'std_person1_birthdate_87D', 'std_person1_empl_employedfrom_271D', 'std_person2_empls_employedfrom_796D', 'std_static_cb_assignmentdate_238D', 'std_static_cb_assignmentdate_4527235D', 'std_static_cb_assignmentdate_4955616D', 'std_static_cb_birthdate_574D', 'std_static_cb_dateofbirth_337D', 'std_static_cb_dateofbirth_342D', 'std_static_cb_responsedate_1012D', 'std_static_cb_responsedate_4527233D', 'std_static_cb_responsedate_4917613D', 'std_tax_a_recorddate_4527225D', 'std_tax_b_deductiondate_4917603D', 'std_tax_c_processingdate_168D', 'mean_collater_typofvalofguarant_298M', 'mean_collater_typofvalofguarant_407M', 'std_collater_typofvalofguarant_407M', 'last_collater_valueofguarantee_1124L', 'last_collater_valueofguarantee_876L', 'mean_collaterals_typeofguarante_359M', 'std_collaterals_typeofguarante_359M', 'mean_collaterals_typeofguarante_669M', 'std_collaterals_typeofguarante_669M', 'last_pmts_dpd_1073P', 'last_pmts_dpd_303P', 'mean_pmts_month_158T', 'mean_pmts_month_706T', 'last_pmts_overdue_1140A', 'last_pmts_overdue_1152A', 'mean_subjectroles_name_541M', 'std_subjectroles_name_541M', 'mean_subjectroles_name_838M', 'std_subjectroles_name_838M', 'last_subjectroles_name_838M', 'count_bureau_b_1_amount_1115A', 'count_bureau_b_1_classificationofcontr_1114M', 'count_bureau_b_1_contractdate_551D', 'count_bureau_b_1_contractmaturitydate_151D', 'count_bureau_b_1_contractst_516M', 'count_bureau_b_1_contracttype_653M', 'count_bureau_b_1_credlmt_1052A', 'count_bureau_b_1_credlmt_228A', 'count_bureau_b_1_credlmt_3940954A', 'count_bureau_b_1_credor_3940957M', 'count_bureau_b_1_credquantity_1099L', 'count_bureau_b_1_credquantity_984L', 'count_bureau_b_1_debtpastduevalue_732A', 'count_bureau_b_1_debtvalue_227A', 'count_bureau_b_1_dpd_550P', 'count_bureau_b_1_dpd_733P', 'count_bureau_b_1_dpdmax_851P', 'count_bureau_b_1_dpdmaxdatemonth_804T', 'count_bureau_b_1_dpdmaxdateyear_742T', 'count_bureau_b_1_installmentamount_644A', 'count_bureau_b_1_installmentamount_833A', 'count_bureau_b_1_instlamount_892A', 'count_bureau_b_1_interesteffectiverate_369L', 'count_bureau_b_1_interestrateyearly_538L', 'count_bureau_b_1_lastupdate_260D', 'count_bureau_b_1_maxdebtpduevalodued_3940955A', 'count_bureau_b_1_num_group1', 'count_bureau_b_1_numberofinstls_810L', 'count_bureau_b_1_overdueamountmax_950A', 'count_bureau_b_1_overdueamountmaxdatemonth_494T', 'count_bureau_b_1_overdueamountmaxdateyear_432T', 'count_bureau_b_1_periodicityofpmts_997L', 'count_bureau_b_1_periodicityofpmts_997M', 'count_bureau_b_1_pmtdaysoverdue_1135P', 'count_bureau_b_1_pmtmethod_731M', 'count_bureau_b_1_pmtnumpending_403L', 'count_bureau_b_1_purposeofcred_722M', 'count_bureau_b_1_residualamount_1093A', 'count_bureau_b_1_residualamount_127A', 'count_bureau_b_1_residualamount_3940956A', 'count_bureau_b_1_subjectrole_326M', 'count_bureau_b_1_subjectrole_43M', 'count_bureau_b_1_totalamount_503A', 'count_bureau_b_1_totalamount_881A', 'count_bureau_b_2_num_group1', 'count_bureau_b_2_num_group2', 'count_bureau_b_2_pmts_date_1107D', 'count_bureau_b_2_pmts_dpdvalue_108P', 'count_bureau_b_2_pmts_pmtsoverdue_635A', 'count_other_amtdebitincoming_4809443A', 'count_other_amtdebitoutgoing_4809440A', 'count_other_amtdepositbalance_4809441A', 'count_other_amtdepositincoming_4809444A', 'count_other_amtdepositoutgoing_4809442A', 'count_other_num_group1', 'count_person1_birth_259D', 'count_person1_incometype_1044T', 'count_person1_mainoccupationinc_384A', 'count_person1_sex_738L', 'count_static_cb_description_5085714M', 'count_static_cb_education_1103M', 'count_static_cb_education_88M', 'count_static_cb_maritalst_385M', 'count_static_cb_maritalst_893M','first_pmts_dpd_303P','first_pmts_overdue_1152A']\n    #这是我在模型训练的时候,一个模型1500迭代器,这些特征只被使用了<5次.\n    useless_cols=['count_applprev1_credacc_transactions_402L', 'last_applprev2_cacccardblochreas_147M', 'last_credit_bureau_a_1_classificationofcontr_13M', 'last_credit_bureau_a_1_contractst_545M', 'count_credit_bureau_a_1_debtoutstand_525A', 'count_credit_bureau_a_1_debtoverdue_47A', 'last_credit_bureau_a_1_description_351M', 'last_credit_bureau_a_1_financialinstitution_591M', 'count_credit_bureau_a_1_overdueamountmax2_14A', 'last_credit_bureau_a_1_purposeofcred_426M', 'last_credit_bureau_a_1_subjectrole_93M', 'count_credit_bureau_a_1_totalamount_6A', 'max_collater_typofvalofguarant_407M', 'last_collater_typofvalofguarant_407M', 'last_collaterals_typeofguarante_359M', 'last_collaterals_typeofguarante_669M', 'last_subjectroles_name_541M', 'count_person1_birthdate_87D', 'count_person1_contaddr_matchlist_1032L', 'count_person1_contaddr_smempladdr_334L', 'count_person1_contaddr_zipcode_807M', 'count_person1_education_927M', 'max_person1_empladdr_district_926M', 'max_person1_empladdr_zipcode_114M', 'count_person1_empladdr_zipcode_114M', 'count_person1_gender_992L', 'count_person1_housingtype_772L', 'count_person1_isreference_387L', 'max_person1_persontype_1072L', 'max_person1_persontype_792L', 'count_person1_persontype_792L', 'count_person1_registaddr_district_1083M', 'count_person1_role_993L', 'count_person1_safeguarantyflag_411L', 'max_person2_conts_role_79M', 'max_person2_empls_economicalst_849M', 'last_person2_empls_economicalst_849M', 'applicationcnt_361L', 'clientscnt_157L', 'clientscnt_257L', 'count_static_cb_days120_123L', 'count_static_cb_days180_256L', 'count_static_cb_days30_165L', 'count_static_cb_days360_512L', 'count_static_cb_days90_310L', 'last_static_cb_education_88M', 'count_static_cb_firstquarter_103L', 'count_static_cb_formonth_118L', 'count_static_cb_formonth_206L', 'count_static_cb_formonth_535L', 'count_static_cb_forquarter_1017L', 'count_static_cb_forquarter_462L', 'count_static_cb_forquarter_634L', 'count_static_cb_fortoday_1092L', 'count_static_cb_forweek_1077L', 'count_static_cb_forweek_528L', 'count_static_cb_forweek_601L', 'count_static_cb_foryear_618L', 'count_static_cb_foryear_818L', 'count_static_cb_foryear_850L', 'count_static_cb_fourthquarter_440L', 'last_static_cb_maritalst_893M', 'count_static_cb_numberofqueries_373L', 'count_static_cb_secondquarter_766L', 'count_static_cb_thirdquarter_1082L', 'count_applprev1_credacc_minhisbal_90A', 'max_credit_bureau_a_1_overdueamount_31A', 'count_credit_bureau_a_1_totaloutstanddebtvalue_39A', 'count_person1_childnum_185L', 'count_person1_empladdr_district_926M', 'count_person1_language1_981M', 'count_person1_personindex_1023L', 'count_person1_registaddr_zipcode_184M', 'count_person1_remitter_829L', 'count_person1_type_25L', 'last_person2_conts_role_79M', 'clientscnt_100L', 'count_applprev1_credacc_maxhisbal_375A', 'count_applprev1_credacc_status_367L', 'count_applprev2_credacc_cards_status_52L', 'applicationscnt_629L', 'numpmtchanneldd_318L', 'count_credit_bureau_a_1_dateofcredend_289D', 'count_credit_bureau_a_1_dateofcredend_353D', 'count_collater_valueofguarantee_1124L', 'count_person1_contaddr_district_15M', 'count_person1_persontype_1072L', 'applications30d_658L', 'clientscnt3m_3712950L', 'clientscnt_304L', 'clientscnt_493L', 'count_deposit_openingdate_313D', 'clientscnt_1130L', 'count_credit_bureau_a_1_interestrate_508L', 'count_credit_bureau_a_1_overdueamountmax_155A', 'count_credit_bureau_a_1_overdueamountmaxdateyear_2T', 'count_person1_maritalst_703L', 'count_person1_num_group1', 'count_static_cb_for3years_584L', 'count_credit_bureau_a_1_refreshdate_3813885D', 'count_credit_bureau_a_1_totaldebtoverduevalue_178A', 'max_person1_education_927M', 'numactivecreds_622L', 'max_applprev1_actualdpd_943P', 'count_applprev1_credacc_actualbalance_314A', 'count_applprev1_revolvingaccount_394A', 'count_credit_bureau_a_1_dateofcredstart_739D', 'count_credit_bureau_a_1_dpdmax_757P', 'count_credit_bureau_a_1_lastupdate_1112D', 'max_credit_bureau_a_1_numberofoverdueinstls_834L', 'max_credit_bureau_a_1_outstandingamount_354A', 'count_credit_bureau_a_1_overdueamountmaxdatemonth_284T', 'count_credit_bureau_a_1_totaldebtoverduevalue_718A', 'max_collaterals_typeofguarante_359M', 'count_debitcard_last180dayturnover_1134A', 'count_debitcard_openingdate_857D', 'count_person1_relationshiptoclient_415T', 'count_person1_role_1084L', 'applicationscnt_464L', 'numactiverelcontr_750L', 'numnotactivated_1143L', 'count_static_cb_dateofbirth_337D', 'max_static_cb_education_88M', 'max_static_cb_maritalst_893M', 'count_static_cb_pmtcount_4527229L', 'count_credit_bureau_a_1_overdueamountmax_35A']\n    \nimport random#提供了一些用于生成随机数的函数\n#设置随机种子,保证模型可以复现\ndef seed_everything(seed):\n    np.random.seed(seed)#numpy的随机种子\n    random.seed(seed)#python内置的随机种子\nseed_everything(Config.seed)\nprint(f\"len(Config.drop_cols):{len(Config.drop_cols)},len(Config.useless_cols):{len(Config.useless_cols)}\")\n\n#读取训练数据中每个特征的dtype\ncolname2dtype=pd.read_csv(\"/kaggle/input/home-credit-inconsistent-data-types/colname2dtype.csv\")\ncolname=colname2dtype['Column'].values\ndtype=colname2dtype['DataType'].values\n\ndtype2pl={}\ndtype2pl['Int64']=pl.Int64\ndtype2pl['Float64']=pl.Float32\ndtype2pl['String']=pl.String\ndtype2pl['Boolean']=pl.String\n\ncolname2dtype={}\nfor idx in range(len(colname)):\n    colname2dtype[colname[idx]]=dtype2pl[dtype[idx]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_agg_feature(feats,df,name='base'):\n    for col in df.drop('case_id').columns:\n        feat=df.group_by('case_id').agg( \n                                        pl.max(col).alias(f\"max_{name}_{col}\"),\n                                        pl.std(col).alias(f\"std_{name}_{col}\"),\n                                        pl.last(col).alias(f\"last_{name}_{col}\"),\n                                        pl.count(col).alias(f\"count_{name}_{col}\"),\n                                        )\n        feats=feats.join(feat,on='case_id',how='left')\n    for col in feats.drop(['case_id']).columns:\n        if (col in Config.drop_cols) or (col in Config.useless_cols):\n            feats=feats.drop([col])\n    return feats\n\n#找到一张表格中每个case_id\ndef find_agg_feats_per_caseid(df):\n    max_feats=pl.DataFrame({\"case_id\":df['case_id'].unique()})\n    for col in df.drop(['case_id']).columns:\n        feat=df.group_by('case_id').agg(  pl.max(col).alias(f\"max_{col}\"),\n                                          pl.mean(col).alias(f\"mean_{col}\"),\n                                          pl.std(col).alias(f\"std_{col}\"),\n                                          pl.first(col).alias(f\"first_{col}\"),\n                                          pl.last(col).alias(f\"last_{col}\"),\n                                          pl.count(col).alias(f\"count_{col}\"),\n                                          pl.n_unique(col).alias(f\"nunique_{col}\"),\n                                       )\n        max_feats=max_feats.join(feat,on='case_id',how='left')\n    for col in max_feats.drop(['case_id']).columns:\n        if (col in Config.drop_cols) or (col in Config.useless_cols):\n            max_feats=max_feats.drop([col])\n    return max_feats\n\ndef set_table_dtypes(df):\n    for col in df.columns:\n        df=df.with_columns(pl.col(col).cast(colname2dtype[col]).alias(col))\n    return df\n\n#遍历表格df的所有列修改数据类型减少内存使用\ndef reduce_mem_usage(df, float16_as32=True):\n    #memory_usage()是df每列的内存使用量,sum是对它们求和, B->KB->MB\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:#遍历每列的列名\n        col_type = df[col].dtype#列名的type\n        if col_type != object and str(col_type)!='category':#不是object也就是说这里处理的是数值类型的变量\n            c_min,c_max = df[col].min(),df[col].max() #求出这列的最大值和最小值\n            if str(col_type)[:3] == 'int':#如果是int类型的变量,不管是int8,int16,int32还是int64\n                #如果这列的取值范围是在int8的取值范围内,那就对类型进行转换 (-128 到 127)\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                #如果这列的取值范围是在int16的取值范围内,那就对类型进行转换(-32,768 到 32,767)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                #如果这列的取值范围是在int32的取值范围内,那就对类型进行转换(-2,147,483,648到2,147,483,647)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                #如果这列的取值范围是在int64的取值范围内,那就对类型进行转换(-9,223,372,036,854,775,808到9,223,372,036,854,775,807)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:#如果是浮点数类型.\n                #如果数值在float16的取值范围内,如果觉得需要更高精度可以考虑float32\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32:#如果数据需要更高的精度可以选择float32\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)  \n                #如果数值在float32的取值范围内，对它进行类型转换\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                #如果数值在float64的取值范围内，对它进行类型转换\n                else:\n                    df[col] = df[col].astype(np.float64)\n    #计算一下结束后的内存\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    #相比一开始的内存减少了百分之多少\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\n#after break 就是仔细研究过文件每个特征含义的意思.\ndef preprocessor(mode='train'):#mode='train'|'test'\n    \n    print(f\"{mode} base file. number:1\")\n    feats=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_base.csv\").pipe(set_table_dtypes)\n    feats=feats.drop(['MONTH'])#'date_decision',\n    \n    print(f\"{mode} applprev_1 file. number:2(3)\")#训练数据2个文件,测试数据3个文件.\n    applprev1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_applprev_1_0.csv\").pipe(set_table_dtypes)\n    file_num=1+int(mode=='test')#测试数据还有2个文件,训练数据还有1个文件,\n    for i in range(file_num):\n        applprev1_i=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_applprev_1_{i+1}.csv\").pipe(set_table_dtypes)\n        applprev1=pl.concat([applprev1,applprev1_i],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n        del applprev1_i\n    feats=find_agg_feature(feats,applprev1,'applprev1')\n    del applprev1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} applprev_2 file. number:1\")\n    applprev2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_applprev_2.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,applprev2,name='applprev2')\n    del applprev2\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} credit_bureau_a_1 file. number:4(5)\")#训练数据4个文件,测试数据5个文件.\n    credit_bureau_a_1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_1_0.csv\").pipe(set_table_dtypes)\n    file_num=3+int(mode=='test')#测试数据还有4个文件,训练数据还有3个文件,\n    for i in range(file_num):\n        credit_bureau_a_1_i=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_1_{i+1}.csv\").pipe(set_table_dtypes)\n        credit_bureau_a_1=pl.concat([credit_bureau_a_1,credit_bureau_a_1_i],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n        del credit_bureau_a_1_i\n        gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    feats=find_agg_feature(feats,credit_bureau_a_1,'credit_bureau_a_1')\n    del credit_bureau_a_1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} credit_bureau_a_2 file. number:11(12)\")#训练数据11个文件,测试数据12个文件.\n    credit_bureau_a_2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_2_0.csv\").pipe(set_table_dtypes)\n    credit_bureau_a_2_max=find_agg_feats_per_caseid(credit_bureau_a_2)\n    file_num=10+int(mode=='test')#测试数据还有10个文件,训练数据还有11个文件,\n    for i in range(file_num):\n        credit_bureau_a_2_i=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_2_{i+1}.csv\").pipe(set_table_dtypes)\n        credit_bureau_a_2_i_max=find_agg_feats_per_caseid(credit_bureau_a_2_i)\n        credit_bureau_a_2_max=pl.concat([credit_bureau_a_2_max,credit_bureau_a_2_i_max],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n        del credit_bureau_a_2_i,credit_bureau_a_2_i_max\n    feats=feats.join(credit_bureau_a_2_max,on='case_id',how='left')\n    del credit_bureau_a_2_max\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} credit_bureau_b file. number:2\")\n    bureau_b_1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_b_1.csv\").pipe(set_table_dtypes)\n    bureau_b_2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_b_2.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,bureau_b_1,name='bureau_b_1')\n    feats=find_agg_feature(feats,bureau_b_2,name='bureau_b_2')\n    del bureau_b_1,bureau_b_2\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n\n    print(f\"{mode} debitcard file. number:1\")\n    debitcard=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_debitcard_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,debitcard,name='debitcard')\n    del debitcard\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} deposit file. number:1\")\n    deposit=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_deposit_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,deposit,name='deposit')\n    del deposit\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} other file. number:1\")\n    other=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_other_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,other,name='other')\n    del other\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} person1 file. number:1\")\n    person1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_person_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,person1,name='person1')   \n    del person1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n\n    print(f\"{mode} person2 file. number:1\")\n    #经过检查person2训练集和测试集对应的列dtype都对应的上\n    person2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_person_2.csv\").pipe(set_table_dtypes)    \n    feats=find_agg_feature(feats,person2,name='person2')\n    del person2\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} static_0 file. number:2(3)\")\n    #pipe用于在DataFrame上自定义自己的函数\n    static_0_0=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_0_0.csv\").pipe(set_table_dtypes)\n    static_0_1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_0_1.csv\").pipe(set_table_dtypes)\n    static=pl.concat([static_0_0,static_0_1],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n    if mode=='test':#如果是测试数据的话还有一个文件\n        static_0_2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_0_2.csv\").pipe(set_table_dtypes)\n        static=pl.concat([static,static_0_2],how=\"vertical_relaxed\")\n    feats=feats.join(static,on='case_id',how='left')\n    del static,static_0_0,static_0_1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} static_cb_file. number:1\")\n    static_cb=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_cb_0.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,static_cb,name='static_cb')\n    del static_cb\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} tax_a file. number:1\")\n    tax_a=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_tax_registry_a_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,tax_a,name='tax_a')\n    del tax_a\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n   \n    print(f\"{mode} tax_b file. number:1\")\n    tax_b=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_tax_registry_b_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,tax_b,name='tax_b')\n    del tax_b\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} tax_c file. number:1\")\n    tax_c=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_tax_registry_c_1.csv\").pipe(set_table_dtypes)\n    if len(tax_c)==0:\n        tax_c=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_tax_registry_c_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,tax_c,name='tax_c')\n    del tax_c\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    print(\"-\"*30)\n    \n    return feats\ntrain_feats=preprocessor(mode='train')\ntrain_feats=train_feats.to_pandas()\ntrain_feats=reduce_mem_usage(train_feats, float16_as32=False)\n\ntest_feats=preprocessor(mode='test')\ntest_feats=test_feats.to_pandas()\ntest_feats=reduce_mem_usage(test_feats, float16_as32=False)\n\ntest_feats.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_cols=[]\nfor col in test_feats:\n    if (col[-1]=='D') and 'count' not in col:\n        date_cols.append(col)\nprint(f\"date_cols:{date_cols}\")\n\n#对日期列变量的处理\ndef deal_date(df):\n    df['date_decision']=pd.to_datetime(df['date_decision'])\n    for col in date_cols:\n        print(f\"col:{col}\")\n        df[col]=pd.to_datetime(df[col])\n        df[f\"{col}_date_decision_gap_day\"]=(df[col]-df['date_decision']).dt.total_seconds() // 86400        \n        df.drop([col],axis=1,inplace=True)\n    df['month_decision'] = df[\"date_decision\"].dt.month\n    df['weekday_decision'] = df[\"date_decision\"].dt.weekday\n    df.drop(['date_decision'],axis=1,inplace=True)\n    return df\ntrain_feats=deal_date(train_feats)\ntest_feats=deal_date(test_feats)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对字符串特征列进行独热编码的转换\nprint(\"----------string one hot encoder ****\")\n            \nfor col in test_feats.columns:\n    if col!='WEEK_NUM':\n        n_unique=train_feats[col].nunique()\n        if n_unique==2 and train_feats[col].dtype=='object':\n            print(f\"one_hot_2:{col}\")\n            unique=train_feats[col].unique()\n            #随便选择一个类别进行转换,比如gender='Female'\n            train_feats[col]=(train_feats[col]==unique[0]).astype(int)\n            test_feats[col]=(test_feats[col]==unique[0]).astype(int)\n        elif n_unique<50 and train_feats[col].dtype=='object':#如果是类别型变量\n            train_feats[col]=(train_feats[col]).astype(\"category\")\n            test_feats[col]=(test_feats[col]).astype(\"category\")\n\nprint(\"----------drop other string or unique value or full null value ****\")\ndrop_cols=[]\nfor col in test_feats.columns:\n    if (train_feats[col].dtype=='object') or (train_feats[col].nunique()==1) or train_feats[col].isna().mean()>0.95:\n        drop_cols+=[col]\nprint(f\"len(drop_cols):{len(drop_cols)},drop_cols:{drop_cols}\")\n#case_id的话就当作普通的id,WEEK_NUM测试数据比训练数据大.\ndrop_cols+=['case_id','WEEK_NUM']\ntrain_feats.drop(drop_cols,axis=1,inplace=True)\ntest_feats.drop(drop_cols,axis=1,inplace=True)\n\ntrain_feats=reduce_mem_usage(train_feats, float16_as32=False)\ntest_feats=reduce_mem_usage(test_feats, float16_as32=False)\n\nprint(f\"len(train_feats):{len(train_feats)},total_features_counts:{len(test_feats.columns)}\")\ntrain_feats.head()\n\n# #数值类型的变量和类别型变量\n# num_cols=[]\n# cat_cols=[]\n# for col in test_feats.columns:\n#     if str(train_feats[col].dtype)!='category':\n#         num_cols.append(col)\n#     else:\n#         cat_cols.append(col)\n# print(f\"len(num_cols):{len(num_cols)},num_cols:{num_cols}\")\n# #根据nan值对col进行分组\n# groupby_nancnt={}\n# for col in num_cols:\n#     nancnt=train_feats[col].isna().sum()#这列缺失值有多少个\n#     #如果是第一个创建列表[col],如果有新的col继续加\n#     try:\n#         groupby_nancnt[nancnt]=[col]\n#     except:\n#         groupby_nancnt[nancnt].append(col)  \n# print(f\"groupby_nancnt:{groupby_nancnt}\")\n\n# def pearson_corr(x1,x2):\n#     \"\"\"\n#     x1,x2:np.array\n#     \"\"\"\n#     mean_x1=np.mean(x1)\n#     mean_x2=np.mean(x2)\n#     std_x1=np.std(x1)\n#     std_x2=np.std(x2)\n#     pearson=np.mean((x1-mean_x1)*(x2-mean_x2))/(std_x1*std_x2)\n#     return pearson\n# #选择特征\n# choose_cols=[]\n# for key,value in groupby_nancnt.items():\n#     if len(value)==1:#如果就是一个特征,那就直接保留就行了\n#         choose_cols+=value\n#     else:#如果缺失值数量=key的列数>1,在相关性大的几列中取nunique最多的一列\n#         remain_cols=value\n#         is_choose=np.zeros(len(remain_cols))#每列是否被选中\n#         for i in range(len(remain_cols)):\n#             groups=[remain_cols[i]]\n#             for j in range(i+1,len(remain_cols)):\n#                 tmp_df=train_feats[remain_cols[i],remain_cols[j]].copy().dropna()\n#                 if abs(pearson_corr(tmp_df[0].values,tmp_df[1].values))>0.8:\n#                     if is_choose[j]==0:\n#                         groups.append(remain_cols[j])\n#                         is_choose[j]=1\n#             max_idx=0;max_unique=1\n#             for idx in range(len(groups)):\n#                 cur_unique=train_feats[groups[idx]].nunique()\n#                 if cur_unique>max_unique():\n#                     max_unique=cur_unique\n#                     max_idx=idx\n#             choose_cols.append(groups[max_idx])\n# choose_cols+=cat_cols\n# print(f\"origin_features_counts:{len(test_feats.columns)},now_features_counts:{len(choose_cols)}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#0405 mean_gini:0.7230855268034984 \nchoose_cols=[col for col in test_feats.columns]\n#保存训练好的树模型,obj是保存的模型,path是需要保存的路径\ndef pickle_dump(obj, path):\n    #打开指定的路径path,binary write(二进制写入)\n    with open(path, mode=\"wb\") as f:\n        #将obj对象保存到f,使用协议版本4进行序列化\n        dill.dump(obj, f, protocol=4)\ndef fit_and_predict(model,train_feats=train_feats,test_feats=test_feats,name=0):\n    X=train_feats[choose_cols].copy()\n    y=train_feats[Config.TARGET_NAME].copy()\n    test_X=test_feats[choose_cols].copy()\n    del train_feats,test_feats\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    oof_pred_pro=np.zeros((len(X),2))\n    test_pred_pro=np.zeros((Config.num_folds,len(test_X),2))\n\n    #10折交叉验证\n    skf = StratifiedKFold(n_splits=Config.num_folds,random_state=Config.seed, shuffle=True)\n\n    for fold, (train_index, valid_index) in (enumerate(skf.split(X, y.astype(str)))):\n        print(f\"fold:{fold}\")\n\n        X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n        y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n        \n        model.fit(X_train,y_train,eval_set=[(X_valid, y_valid)],\n                  callbacks=[log_evaluation(100),early_stopping(100)]\n                 )\n        \n        oof_pred_pro[valid_index]=model.predict_proba(X_valid)\n        #将数据分批次进行预测.\n        for idx in range(0,len(test_X),Config.batch_size):\n            test_pred_pro[fold][idx:idx+Config.batch_size]=model.predict_proba(test_X[idx:idx+Config.batch_size])\n        pickle_dump(model, f'/kaggle/working/lgb_fold{Config.num_folds*name+fold}.model') #保存训练好的模型   \n    gini=2*roc_auc_score(y.values,oof_pred_pro[:,1])-1\n    print(f\"mean_gini:{gini}\")\n    \n    return oof_pred_pro,test_pred_pro\n\n#参数来源:https://www.kaggle.com/code/daviddirethucus/home-credit-risk-lightgbm\nlgb_params={\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2560,\n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": Config.seed,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\":True,\n    'num_leaves':64,\n    \"verbose\": -1,\n    \"max_bin\":245,\n    'device':'gpu',\n    }\n\nlgb_oof_pred_pro,lgb_test_pred_pro=fit_and_predict(model= LGBMClassifier(**lgb_params),\n                                                  train_feats=train_feats,test_feats=test_feats,name=0\n                                                  )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds=lgb_test_pred_pro.mean(axis=0)[:,1]\nsubmission=pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nsubmission['score']=np.clip(np.nan_to_num(test_preds,nan=0.314),0,1)\nsubmission.to_csv(\"submission.csv\",index=None)\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#这里就是个示例代码,有很多特征,然后还有每折交叉验证的特征重要性,最后得出小于margin的不重要的特征.\n# import re\n# import numpy as np\n# total_features=['a','b','c']\n# text=\"\"\"[1 22 23] [3 2 30]\n# \"\"\"\n# nums=re.findall(r'\\d+',text)\n# nums=np.array([int(num) for num in nums])\n# margin=5#如果一个特征在2000个迭代器中选择的次数少于5次,就说明这个特征不重要.\n# useless_cols=[]\n# for i in range(len(nums)):\n#     if nums[i]<margin:\n#         feats=total_features[i%len(total_features)]\n#         if feats not in useless_cols:\n#             useless_cols.append(feats)\n# print(f\"len(useless_cols):{len(useless_cols)},useless_cols:{useless_cols}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Future work to be done:\n\n### Selection and construction of features\n\n### ensemble more models\n\n### ……","metadata":{}}]}