{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":166996856,"sourceType":"kernelVersion"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl  # 파이썬에서 대규모 데이터 세트를 처리하는 데 더 효율적인 성능을 제공하는 라이브러리입니다.\nimport pandas as pd  # CSV 파일을 가져오는 데 사용하는 라이브러리입니다.\nimport numpy as np  # 행렬 연산을 수행하는 데 사용되는 라이브러리입니다.\nfrom lightgbm import LGBMClassifier, log_evaluation, early_stopping  # 분류 모델인 LGBMClassifier, 로그 평가, 과적합 방지를 위한 조기 중지를 수행하는 함수입니다.\nfrom sklearn.metrics import roc_auc_score  # ROC AUC 곡선을 가져오는 데 사용되는 라이브러리입니다.\nfrom sklearn.model_selection import StratifiedKFold  # KFold는 단순히 k 폴드로 분할하는 반면, StratifiedKFold는 각 클래스 비율을 고려합니다.\nimport dill  # 객체를 직렬화하고 역직렬화하는 데 사용되는 라이브러리입니다. (예: 트리 모델 저장 및 로드)\nimport gc  # 가비지 컬렉션 모듈입니다. (메모리 관리)\nimport time  # 표준 라이브러리의 시간 모듈입니다.\n\nimport warnings\n# 경고 메시지 출력 비활성화\nwarnings.filterwarnings(\"ignore\")\n\n# 노트북의 학습 시간을 제공하기 위해 모델 트레이닝 시간을 제공합니다.\n# time.strftime() 함수는 시간 객체를 문자열로 형식화합니다. time.localtime() 함수는 현재 로컬 시간을 나타내는 time.struct_time 객체를 반환합니다.\ncurrent_time = time.strftime(\"%Y-%m-%d %H:%M:%S\", time.localtime())\nprint(\"this notebook training time is \", current_time)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-14T00:30:19.001165Z","iopub.execute_input":"2024-04-14T00:30:19.002121Z","iopub.status.idle":"2024-04-14T00:30:21.796042Z","shell.execute_reply.started":"2024-04-14T00:30:19.002085Z","shell.execute_reply":"2024-04-14T00:30:21.795065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#config\nclass Config():\n    seed=2024 # 랜덤 시드값\n    num_folds=10 # 교차 검증에서의 fold 수\n    TARGET_NAME ='target' # 타겟 변수의 이름\n    batch_size=1000 # 테스트 데이터의 크기를 모르므로, 모델에 배치로 데이터를 넣음.\n    # 비범주형 문자열, nunique=1, 결측치 비율이 95% 이상인 열\n    # 날짜 관련 열은 현재 date_decision과의 gap_day를 계산했고, gap_day를 drop할 예정이라면, 원래 날짜 열을 바로 drop하는 것이 좋음.\n    drop_cols=['max_applprev1_cancelreason_3545846M', 'last_applprev1_cancelreason_3545846M', 'max_applprev1_district_544M', 'last_applprev1_district_544M', 'max_applprev1_isbidproduct_390L', 'last_applprev1_isbidproduct_390L', 'max_applprev1_isdebitcard_527L', 'last_applprev1_isdebitcard_527L', 'max_applprev1_profession_152M', 'last_applprev1_profession_152M', 'last_applprev1_revolvingaccount_394A', 'last_applprev2_credacc_cards_status_52L', 'last_credit_bureau_a_1_annualeffectiverate_199L', 'last_credit_bureau_a_1_annualeffectiverate_63L', 'max_credit_bureau_a_1_classificationofcontr_400M', 'last_credit_bureau_a_1_classificationofcontr_400M', 'max_credit_bureau_a_1_contractst_964M', 'last_credit_bureau_a_1_contractst_964M', 'last_credit_bureau_a_1_contractsum_5085717L', 'last_credit_bureau_a_1_credlmt_230A', 'last_credit_bureau_a_1_credlmt_935A', 'last_credit_bureau_a_1_debtoutstand_525A', 'last_credit_bureau_a_1_debtoverdue_47A', 'last_credit_bureau_a_1_dpdmax_139P', 'last_credit_bureau_a_1_dpdmax_757P', 'last_credit_bureau_a_1_dpdmaxdatemonth_442T', 'last_credit_bureau_a_1_dpdmaxdatemonth_89T', 'last_credit_bureau_a_1_dpdmaxdateyear_596T', 'last_credit_bureau_a_1_dpdmaxdateyear_896T', 'max_credit_bureau_a_1_financialinstitution_382M', 'last_credit_bureau_a_1_financialinstitution_382M', 'max_credit_bureau_a_1_financialinstitution_591M', 'last_credit_bureau_a_1_instlamount_768A', 'last_credit_bureau_a_1_instlamount_852A', 'max_credit_bureau_a_1_interestrate_508L', 'last_credit_bureau_a_1_interestrate_508L', 'last_credit_bureau_a_1_monthlyinstlamount_332A', 'last_credit_bureau_a_1_monthlyinstlamount_674A', 'last_credit_bureau_a_1_nominalrate_281L', 'last_credit_bureau_a_1_nominalrate_498L', 'last_credit_bureau_a_1_numberofcontrsvalue_258L', 'last_credit_bureau_a_1_numberofcontrsvalue_358L', 'last_credit_bureau_a_1_numberofinstls_229L', 'last_credit_bureau_a_1_numberofinstls_320L', 'last_credit_bureau_a_1_numberofoutstandinstls_520L', 'last_credit_bureau_a_1_numberofoutstandinstls_59L', 'last_credit_bureau_a_1_numberofoverdueinstlmax_1039L', 'last_credit_bureau_a_1_numberofoverdueinstlmax_1151L', 'last_credit_bureau_a_1_numberofoverdueinstls_725L', 'last_credit_bureau_a_1_numberofoverdueinstls_834L', 'last_credit_bureau_a_1_outstandingamount_354A', 'last_credit_bureau_a_1_outstandingamount_362A', 'last_credit_bureau_a_1_overdueamount_31A', 'last_credit_bureau_a_1_overdueamount_659A', 'last_credit_bureau_a_1_overdueamountmax2_14A', 'last_credit_bureau_a_1_overdueamountmax2_398A', 'last_credit_bureau_a_1_overdueamountmax_155A', 'last_credit_bureau_a_1_overdueamountmax_35A', 'last_credit_bureau_a_1_overdueamountmaxdatemonth_284T', 'last_credit_bureau_a_1_overdueamountmaxdatemonth_365T', 'last_credit_bureau_a_1_overdueamountmaxdateyear_2T', 'last_credit_bureau_a_1_overdueamountmaxdateyear_994T', 'last_credit_bureau_a_1_periodicityofpmts_1102L', 'last_credit_bureau_a_1_periodicityofpmts_837L', 'last_credit_bureau_a_1_prolongationcount_1120L', 'max_credit_bureau_a_1_prolongationcount_599L', 'last_credit_bureau_a_1_prolongationcount_599L', 'last_credit_bureau_a_1_residualamount_488A', 'last_credit_bureau_a_1_residualamount_856A', 'last_credit_bureau_a_1_subjectrole_182M', 'last_credit_bureau_a_1_totalamount_6A', 'last_credit_bureau_a_1_totalamount_996A', 'last_credit_bureau_a_1_totaldebtoverduevalue_178A', 'last_credit_bureau_a_1_totaldebtoverduevalue_718A', 'last_credit_bureau_a_1_totaloutstanddebtvalue_39A', 'last_credit_bureau_a_1_totaloutstanddebtvalue_668A', 'max_collater_typofvalofguarant_298M', 'last_collater_typofvalofguarant_298M', 'std_collater_typofvalofguarant_298M', 'max_pmts_month_158T', 'last_pmts_month_158T', 'std_pmts_month_158T', 'max_pmts_month_706T', 'last_pmts_month_706T', 'std_pmts_month_706T', 'max_bureau_b_1_amount_1115A', 'last_bureau_b_1_amount_1115A', 'max_bureau_b_1_classificationofcontr_1114M', 'last_bureau_b_1_classificationofcontr_1114M', 'max_bureau_b_1_contractst_516M', 'last_bureau_b_1_contractst_516M', 'max_bureau_b_1_contracttype_653M', 'last_bureau_b_1_contracttype_653M', 'max_bureau_b_1_credlmt_1052A', 'last_bureau_b_1_credlmt_1052A', 'max_bureau_b_1_credlmt_228A', 'last_bureau_b_1_credlmt_228A', 'max_bureau_b_1_credlmt_3940954A', 'last_bureau_b_1_credlmt_3940954A', 'max_bureau_b_1_credor_3940957M', 'last_bureau_b_1_credor_3940957M', 'max_bureau_b_1_credquantity_1099L', 'last_bureau_b_1_credquantity_1099L', 'max_bureau_b_1_credquantity_984L', 'last_bureau_b_1_credquantity_984L', 'max_bureau_b_1_debtpastduevalue_732A', 'last_bureau_b_1_debtpastduevalue_732A', 'max_bureau_b_1_debtvalue_227A', 'last_bureau_b_1_debtvalue_227A', 'max_bureau_b_1_dpd_550P', 'last_bureau_b_1_dpd_550P', 'max_bureau_b_1_dpd_733P', 'last_bureau_b_1_dpd_733P', 'max_bureau_b_1_dpdmax_851P', 'last_bureau_b_1_dpdmax_851P', 'max_bureau_b_1_dpdmaxdatemonth_804T', 'last_bureau_b_1_dpdmaxdatemonth_804T', 'max_bureau_b_1_dpdmaxdateyear_742T', 'last_bureau_b_1_dpdmaxdateyear_742T', 'max_bureau_b_1_installmentamount_644A', 'last_bureau_b_1_installmentamount_644A', 'max_bureau_b_1_installmentamount_833A', 'last_bureau_b_1_installmentamount_833A', 'max_bureau_b_1_instlamount_892A', 'last_bureau_b_1_instlamount_892A', 'max_bureau_b_1_interesteffectiverate_369L', 'last_bureau_b_1_interesteffectiverate_369L', 'max_bureau_b_1_interestrateyearly_538L', 'last_bureau_b_1_interestrateyearly_538L', 'max_bureau_b_1_maxdebtpduevalodued_3940955A', 'last_bureau_b_1_maxdebtpduevalodued_3940955A', 'max_bureau_b_1_num_group1', 'last_bureau_b_1_num_group1', 'max_bureau_b_1_numberofinstls_810L', 'last_bureau_b_1_numberofinstls_810L', 'max_bureau_b_1_overdueamountmax_950A', 'last_bureau_b_1_overdueamountmax_950A', 'max_bureau_b_1_overdueamountmaxdatemonth_494T', 'last_bureau_b_1_overdueamountmaxdatemonth_494T', 'max_bureau_b_1_overdueamountmaxdateyear_432T', 'last_bureau_b_1_overdueamountmaxdateyear_432T', 'max_bureau_b_1_periodicityofpmts_997L', 'last_bureau_b_1_periodicityofpmts_997L', 'max_bureau_b_1_periodicityofpmts_997M', 'last_bureau_b_1_periodicityofpmts_997M', 'max_bureau_b_1_pmtdaysoverdue_1135P', 'last_bureau_b_1_pmtdaysoverdue_1135P', 'max_bureau_b_1_pmtmethod_731M', 'last_bureau_b_1_pmtmethod_731M', 'max_bureau_b_1_pmtnumpending_403L', 'last_bureau_b_1_pmtnumpending_403L', 'max_bureau_b_1_purposeofcred_722M', 'last_bureau_b_1_purposeofcred_722M', 'max_bureau_b_1_residualamount_1093A', 'last_bureau_b_1_residualamount_1093A', 'max_bureau_b_1_residualamount_127A', 'last_bureau_b_1_residualamount_127A', 'max_bureau_b_1_residualamount_3940956A', 'last_bureau_b_1_residualamount_3940956A', 'max_bureau_b_1_subjectrole_326M', 'last_bureau_b_1_subjectrole_326M', 'max_bureau_b_1_subjectrole_43M', 'last_bureau_b_1_subjectrole_43M', 'max_bureau_b_1_totalamount_503A', 'last_bureau_b_1_totalamount_503A', 'max_bureau_b_1_totalamount_881A', 'last_bureau_b_1_totalamount_881A', 'max_bureau_b_2_num_group1', 'last_bureau_b_2_num_group1', 'max_bureau_b_2_num_group2', 'last_bureau_b_2_num_group2', 'max_bureau_b_2_pmts_dpdvalue_108P', 'last_bureau_b_2_pmts_dpdvalue_108P', 'max_bureau_b_2_pmts_pmtsoverdue_635A', 'last_bureau_b_2_pmts_pmtsoverdue_635A', 'max_debitcard_last180dayaveragebalance_704A', 'last_debitcard_last180dayaveragebalance_704A', 'max_debitcard_last180dayturnover_1134A', 'last_debitcard_last180dayturnover_1134A', 'max_debitcard_last30dayturnover_651A', 'last_debitcard_last30dayturnover_651A', 'max_other_amtdebitincoming_4809443A', 'last_other_amtdebitincoming_4809443A', 'max_other_amtdebitoutgoing_4809440A', 'last_other_amtdebitoutgoing_4809440A', 'max_other_amtdepositbalance_4809441A', 'last_other_amtdepositbalance_4809441A', 'max_other_amtdepositincoming_4809444A', 'last_other_amtdepositincoming_4809444A', 'max_other_amtdepositoutgoing_4809442A', 'last_other_amtdepositoutgoing_4809442A', 'max_other_num_group1', 'last_other_num_group1', 'max_person1_childnum_185L', 'last_person1_childnum_185L', 'max_person1_contaddr_district_15M', 'last_person1_contaddr_district_15M', 'max_person1_contaddr_matchlist_1032L', 'last_person1_contaddr_matchlist_1032L', 'last_person1_contaddr_smempladdr_334L', 'max_person1_contaddr_zipcode_807M', 'last_person1_contaddr_zipcode_807M', 'last_person1_empl_employedtotal_800L', 'last_person1_empl_industry_691L', 'last_person1_empladdr_district_926M', 'last_person1_empladdr_zipcode_114M', 'last_person1_familystate_447L', 'max_person1_gender_992L', 'last_person1_gender_992L', 'last_person1_housetype_905L', 'max_person1_housingtype_772L', 'last_person1_housingtype_772L', 'max_person1_isreference_387L', 'last_person1_isreference_387L', 'max_person1_maritalst_703L', 'last_person1_maritalst_703L', 'max_person1_registaddr_district_1083M', 'last_person1_registaddr_district_1083M', 'max_person1_registaddr_zipcode_184M', 'last_person1_registaddr_zipcode_184M', 'max_person1_remitter_829L', 'last_person1_remitter_829L', 'max_person1_role_993L', 'last_person1_role_993L', 'last_person1_safeguarantyflag_411L', 'last_person1_sex_738L', 'max_person2_addres_district_368M', 'last_person2_addres_district_368M', 'max_person2_addres_role_871L', 'last_person2_addres_role_871L', 'max_person2_addres_zip_823M', 'last_person2_addres_zip_823M', 'max_person2_empls_employer_name_740M', 'last_person2_empls_employer_name_740M', 'max_person2_relatedpersons_role_762T', 'last_person2_relatedpersons_role_762T', 'amtinstpaidbefduel24m_4187115A', 'avgdbddpdlast3m_4187120P', 'avgdbdtollast24m_4525197P', 'avglnamtstart24m_4525187A', 'avgoutstandbalancel6m_4187114A', 'avgpmtlast12m_4525200A', 'bankacctype_710L', 'cardtype_51L', 'clientscnt_136L', 'commnoinclast6m_3546845L', 'deferredmnthsnum_166L', 'equalitydataagreement_891L', 'equalityempfrom_62L', 'interestrategrace_34L', 'isbidproductrequest_292L', 'isdebitcard_729L', 'lastapprcommoditytypec_5251766M', 'lastcancelreason_561M', 'lastdependentsnum_448L', 'lastotherinc_902A', 'lastotherlnsexpense_631A', 'lastrejectcommodtypec_5251769M', 'mastercontrelectronic_519L', 'mastercontrexist_109L', 'maxannuity_4075009A', 'maxdbddpdtollast6m_4187119P', 'maxlnamtstart6m_4525199A', 'maxoutstandbalancel12m_4187113A', 'maxpmtlast3m_4525190A', 'mindbdtollast24m_4525191P', 'numinstlswithdpd5_4187116L', 'numinstmatpaidtearly2d_4499204L', 'numinstpaid_4499208L', 'numinstpaidearly3dest_4493216L', 'numinstpaidearly5dest_4493211L', 'numinstpaidearly5dobd_4499205L', 'numinstpaidearlyest_4493214L', 'numinstpaidlastcontr_4325080L', 'numinstregularpaidest_4493210L', 'numinsttopaygrest_4493213L', 'numinstunpaidmaxest_4493212L', 'opencred_647L', 'paytype1st_925L', 'paytype_783L', 'previouscontdistrict_112M', 'sumoutstandtotalest_4493215A', 'totinstallast1m_4525188A', 'typesuite_864L', 'max_static_cb_contractssum_5085716L', 'last_static_cb_contractssum_5085716L', 'max_static_cb_description_5085714M', 'last_static_cb_description_5085714M', 'max_static_cb_for3years_128L', 'last_static_cb_for3years_128L', 'max_static_cb_for3years_504L', 'last_static_cb_for3years_504L', 'max_static_cb_for3years_584L', 'last_static_cb_for3years_584L', 'max_static_cb_formonth_118L', 'last_static_cb_formonth_118L', 'max_static_cb_formonth_206L', 'last_static_cb_formonth_206L', 'max_static_cb_formonth_535L', 'last_static_cb_formonth_535L', 'max_static_cb_forquarter_1017L', 'last_static_cb_forquarter_1017L', 'max_static_cb_forquarter_462L', 'last_static_cb_forquarter_462L', 'max_static_cb_forquarter_634L', 'last_static_cb_forquarter_634L', 'max_static_cb_fortoday_1092L', 'last_static_cb_fortoday_1092L', 'max_static_cb_forweek_1077L', 'last_static_cb_forweek_1077L', 'max_static_cb_forweek_528L', 'last_static_cb_forweek_528L', 'max_static_cb_forweek_601L', 'last_static_cb_forweek_601L', 'max_static_cb_foryear_618L', 'last_static_cb_foryear_618L', 'max_static_cb_foryear_818L', 'last_static_cb_foryear_818L', 'max_static_cb_foryear_850L', 'last_static_cb_foryear_850L', 'max_static_cb_pmtaverage_4527227A', 'last_static_cb_pmtaverage_4527227A', 'max_static_cb_pmtaverage_4955615A', 'last_static_cb_pmtaverage_4955615A', 'max_static_cb_pmtcount_4955617L', 'last_static_cb_pmtcount_4955617L', 'max_static_cb_riskassesment_302T', 'last_static_cb_riskassesment_302T', 'max_static_cb_riskassesment_940T', 'last_static_cb_riskassesment_940T', 'max_tax_a_name_4527232M', 'last_tax_a_name_4527232M', 'max_tax_b_name_4917606M', 'last_tax_b_name_4917606M', 'max_tax_c_employername_160M', 'last_tax_c_employername_160M', 'last_credit_bureau_a_1_dateofcredend_289D', 'last_credit_bureau_a_1_dateofcredend_353D', 'last_credit_bureau_a_1_dateofcredstart_181D', 'last_credit_bureau_a_1_dateofcredstart_739D', 'last_credit_bureau_a_1_dateofrealrepmt_138D', 'last_credit_bureau_a_1_lastupdate_1112D', 'last_credit_bureau_a_1_lastupdate_388D', 'last_credit_bureau_a_1_numberofoverdueinstlmaxdat_148D', 'last_credit_bureau_a_1_numberofoverdueinstlmaxdat_641D', 'last_credit_bureau_a_1_overdueamountmax2date_1002D', 'last_credit_bureau_a_1_overdueamountmax2date_1142D', 'max_bureau_b_1_contractdate_551D', 'last_bureau_b_1_contractdate_551D', 'max_bureau_b_1_contractmaturitydate_151D', 'last_bureau_b_1_contractmaturitydate_151D', 'max_bureau_b_1_lastupdate_260D', 'last_bureau_b_1_lastupdate_260D', 'max_bureau_b_2_pmts_date_1107D', 'last_bureau_b_2_pmts_date_1107D', 'max_deposit_contractenddate_991D', 'last_deposit_contractenddate_991D', 'max_person1_birthdate_87D', 'last_person1_birthdate_87D', 'last_person1_empl_employedfrom_271D', 'max_person2_empls_employedfrom_796D', 'last_person2_empls_employedfrom_796D', 'lastrepayingdate_696D', 'payvacationpostpone_4187118D', 'max_static_cb_assignmentdate_4955616D', 'last_static_cb_assignmentdate_4955616D', 'max_static_cb_dateofbirth_342D', 'last_static_cb_dateofbirth_342D', 'std_applprev1_cancelreason_3545846M', 'std_applprev1_credacc_actualbalance_314A', 'std_applprev1_credacc_maxhisbal_375A', 'std_applprev1_credacc_minhisbal_90A', 'std_applprev1_credacc_status_367L', 'std_applprev1_credacc_transactions_402L', 'std_applprev1_credtype_587L', 'std_applprev1_district_544M', 'std_applprev1_education_1138M', 'std_applprev1_familystate_726L', 'std_applprev1_inittransactioncode_279L', 'std_applprev1_isbidproduct_390L', 'std_applprev1_isdebitcard_527L', 'std_applprev1_postype_4733339M', 'std_applprev1_profession_152M', 'std_applprev1_rejectreason_755M', 'std_applprev1_rejectreasonclient_4145042M', 'std_applprev1_revolvingaccount_394A', 'std_applprev1_status_219L', 'std_applprev2_cacccardblochreas_147M', 'std_applprev2_conts_type_509L', 'std_applprev2_credacc_cards_status_52L', 'std_credit_bureau_a_1_annualeffectiverate_63L', 'std_credit_bureau_a_1_classificationofcontr_13M', 'std_credit_bureau_a_1_classificationofcontr_400M', 'std_credit_bureau_a_1_contractst_545M', 'std_credit_bureau_a_1_contractst_964M', 'std_credit_bureau_a_1_debtoutstand_525A', 'std_credit_bureau_a_1_debtoverdue_47A', 'std_credit_bureau_a_1_description_351M', 'std_credit_bureau_a_1_financialinstitution_382M', 'std_credit_bureau_a_1_financialinstitution_591M', 'std_credit_bureau_a_1_interestrate_508L', 'std_credit_bureau_a_1_numberofcontrsvalue_258L', 'std_credit_bureau_a_1_numberofcontrsvalue_358L', 'std_credit_bureau_a_1_prolongationcount_1120L', 'std_credit_bureau_a_1_prolongationcount_599L', 'std_credit_bureau_a_1_purposeofcred_426M', 'std_credit_bureau_a_1_purposeofcred_874M', 'std_credit_bureau_a_1_subjectrole_182M', 'std_credit_bureau_a_1_subjectrole_93M', 'std_credit_bureau_a_1_totaldebtoverduevalue_178A', 'std_credit_bureau_a_1_totaldebtoverduevalue_718A', 'std_credit_bureau_a_1_totaloutstanddebtvalue_39A', 'std_credit_bureau_a_1_totaloutstanddebtvalue_668A', 'std_bureau_b_1_amount_1115A', 'std_bureau_b_1_classificationofcontr_1114M', 'std_bureau_b_1_contractst_516M', 'std_bureau_b_1_contracttype_653M', 'std_bureau_b_1_credlmt_1052A', 'std_bureau_b_1_credlmt_228A', 'std_bureau_b_1_credlmt_3940954A', 'std_bureau_b_1_credor_3940957M', 'std_bureau_b_1_credquantity_1099L', 'std_bureau_b_1_credquantity_984L', 'std_bureau_b_1_debtpastduevalue_732A', 'std_bureau_b_1_debtvalue_227A', 'std_bureau_b_1_dpd_550P', 'std_bureau_b_1_dpd_733P', 'std_bureau_b_1_dpdmax_851P', 'std_bureau_b_1_dpdmaxdatemonth_804T', 'std_bureau_b_1_dpdmaxdateyear_742T', 'std_bureau_b_1_installmentamount_644A', 'std_bureau_b_1_installmentamount_833A', 'std_bureau_b_1_instlamount_892A', 'std_bureau_b_1_interesteffectiverate_369L', 'std_bureau_b_1_interestrateyearly_538L', 'std_bureau_b_1_maxdebtpduevalodued_3940955A', 'std_bureau_b_1_num_group1', 'std_bureau_b_1_numberofinstls_810L', 'std_bureau_b_1_overdueamountmax_950A', 'std_bureau_b_1_overdueamountmaxdatemonth_494T', 'std_bureau_b_1_overdueamountmaxdateyear_432T', 'std_bureau_b_1_periodicityofpmts_997L', 'std_bureau_b_1_periodicityofpmts_997M', 'std_bureau_b_1_pmtdaysoverdue_1135P', 'std_bureau_b_1_pmtmethod_731M', 'std_bureau_b_1_pmtnumpending_403L', 'std_bureau_b_1_purposeofcred_722M', 'std_bureau_b_1_residualamount_1093A', 'std_bureau_b_1_residualamount_127A', 'std_bureau_b_1_residualamount_3940956A', 'std_bureau_b_1_subjectrole_326M', 'std_bureau_b_1_subjectrole_43M', 'std_bureau_b_1_totalamount_503A', 'std_bureau_b_1_totalamount_881A', 'std_bureau_b_2_num_group1', 'std_bureau_b_2_num_group2', 'std_bureau_b_2_pmts_dpdvalue_108P', 'std_bureau_b_2_pmts_pmtsoverdue_635A', 'std_debitcard_last180dayaveragebalance_704A', 'std_debitcard_last180dayturnover_1134A', 'std_debitcard_last30dayturnover_651A', 'std_debitcard_num_group1', 'std_deposit_amount_416A', 'std_deposit_num_group1', 'std_other_amtdebitincoming_4809443A', 'std_other_amtdebitoutgoing_4809440A', 'std_other_amtdepositbalance_4809441A', 'std_other_amtdepositincoming_4809444A', 'std_other_amtdepositoutgoing_4809442A', 'std_other_num_group1', 'std_person1_childnum_185L', 'std_person1_contaddr_district_15M', 'std_person1_contaddr_matchlist_1032L', 'std_person1_contaddr_smempladdr_334L', 'std_person1_contaddr_zipcode_807M', 'std_person1_education_927M', 'std_person1_empl_employedtotal_800L', 'std_person1_empl_industry_691L', 'std_person1_empladdr_district_926M', 'std_person1_empladdr_zipcode_114M', 'std_person1_familystate_447L', 'std_person1_gender_992L', 'std_person1_housetype_905L', 'std_person1_housingtype_772L', 'std_person1_incometype_1044T', 'std_person1_isreference_387L', 'std_person1_language1_981M', 'std_person1_mainoccupationinc_384A', 'std_person1_maritalst_703L', 'std_person1_registaddr_district_1083M', 'std_person1_registaddr_zipcode_184M', 'std_person1_relationshiptoclient_415T', 'std_person1_relationshiptoclient_642T', 'std_person1_remitter_829L', 'std_person1_role_1084L', 'std_person1_role_993L', 'std_person1_safeguarantyflag_411L', 'std_person1_sex_738L', 'std_person1_type_25L', 'std_person2_addres_district_368M', 'std_person2_addres_role_871L', 'std_person2_addres_zip_823M', 'std_person2_conts_role_79M', 'std_person2_empls_economicalst_849M', 'std_person2_empls_employer_name_740M', 'std_person2_relatedpersons_role_762T', 'std_static_cb_contractssum_5085716L', 'std_static_cb_days120_123L', 'std_static_cb_days180_256L', 'std_static_cb_days30_165L', 'std_static_cb_days360_512L', 'std_static_cb_days90_310L', 'std_static_cb_description_5085714M', 'std_static_cb_education_1103M', 'std_static_cb_education_88M', 'std_static_cb_firstquarter_103L', 'std_static_cb_for3years_128L', 'std_static_cb_for3years_504L', 'std_static_cb_for3years_584L', 'std_static_cb_formonth_118L', 'std_static_cb_formonth_206L', 'std_static_cb_formonth_535L', 'std_static_cb_forquarter_1017L', 'std_static_cb_forquarter_462L', 'std_static_cb_forquarter_634L', 'std_static_cb_fortoday_1092L', 'std_static_cb_forweek_1077L', 'std_static_cb_forweek_528L', 'std_static_cb_forweek_601L', 'std_static_cb_foryear_618L', 'std_static_cb_foryear_818L', 'std_static_cb_foryear_850L', 'std_static_cb_fourthquarter_440L', 'std_static_cb_maritalst_385M', 'std_static_cb_maritalst_893M', 'std_static_cb_numberofqueries_373L', 'std_static_cb_pmtaverage_3A', 'std_static_cb_pmtaverage_4527227A', 'std_static_cb_pmtaverage_4955615A', 'std_static_cb_pmtcount_4527229L', 'std_static_cb_pmtcount_4955617L', 'std_static_cb_pmtcount_693L', 'std_static_cb_pmtscount_423L', 'std_static_cb_pmtssum_45A', 'std_static_cb_requesttype_4525192L', 'std_static_cb_riskassesment_302T', 'std_static_cb_riskassesment_940T', 'std_static_cb_secondquarter_766L', 'std_static_cb_thirdquarter_1082L', 'std_tax_a_name_4527232M', 'std_tax_b_name_4917606M', 'std_tax_c_employername_160M', 'std_applprev1_approvaldate_319D', 'std_applprev1_creationdate_885D', 'std_applprev1_dateactivated_425D', 'std_applprev1_dtlastpmt_581D', 'std_applprev1_dtlastpmtallstes_3545839D', 'std_applprev1_employedfrom_700D', 'std_applprev1_firstnonzeroinstldate_307D', 'std_credit_bureau_a_1_dateofcredend_289D', 'std_credit_bureau_a_1_dateofcredend_353D', 'std_credit_bureau_a_1_dateofcredstart_181D', 'std_credit_bureau_a_1_dateofcredstart_739D', 'std_credit_bureau_a_1_dateofrealrepmt_138D', 'std_credit_bureau_a_1_lastupdate_1112D', 'std_credit_bureau_a_1_lastupdate_388D', 'std_credit_bureau_a_1_numberofoverdueinstlmaxdat_148D', 'std_credit_bureau_a_1_numberofoverdueinstlmaxdat_641D', 'std_credit_bureau_a_1_overdueamountmax2date_1002D', 'std_credit_bureau_a_1_overdueamountmax2date_1142D', 'std_credit_bureau_a_1_refreshdate_3813885D', 'std_bureau_b_1_contractdate_551D', 'std_bureau_b_1_contractmaturitydate_151D', 'std_bureau_b_1_lastupdate_260D', 'std_bureau_b_2_pmts_date_1107D', 'std_debitcard_openingdate_857D', 'std_deposit_contractenddate_991D', 'std_deposit_openingdate_313D', 'std_person1_birth_259D', 'std_person1_birthdate_87D', 'std_person1_empl_employedfrom_271D', 'std_person2_empls_employedfrom_796D', 'std_static_cb_assignmentdate_238D', 'std_static_cb_assignmentdate_4527235D', 'std_static_cb_assignmentdate_4955616D', 'std_static_cb_birthdate_574D', 'std_static_cb_dateofbirth_337D', 'std_static_cb_dateofbirth_342D', 'std_static_cb_responsedate_1012D', 'std_static_cb_responsedate_4527233D', 'std_static_cb_responsedate_4917613D', 'std_tax_a_recorddate_4527225D', 'std_tax_b_deductiondate_4917603D', 'std_tax_c_processingdate_168D', 'mean_collater_typofvalofguarant_298M', 'mean_collater_typofvalofguarant_407M', 'std_collater_typofvalofguarant_407M', 'last_collater_valueofguarantee_1124L', 'last_collater_valueofguarantee_876L', 'mean_collaterals_typeofguarante_359M', 'std_collaterals_typeofguarante_359M', 'mean_collaterals_typeofguarante_669M', 'std_collaterals_typeofguarante_669M', 'last_pmts_dpd_1073P', 'last_pmts_dpd_303P', 'mean_pmts_month_158T', 'mean_pmts_month_706T', 'last_pmts_overdue_1140A', 'last_pmts_overdue_1152A', 'mean_subjectroles_name_541M', 'std_subjectroles_name_541M', 'mean_subjectroles_name_838M', 'std_subjectroles_name_838M', 'last_subjectroles_name_838M', 'count_bureau_b_1_amount_1115A', 'count_bureau_b_1_classificationofcontr_1114M', 'count_bureau_b_1_contractdate_551D', 'count_bureau_b_1_contractmaturitydate_151D', 'count_bureau_b_1_contractst_516M', 'count_bureau_b_1_contracttype_653M', 'count_bureau_b_1_credlmt_1052A', 'count_bureau_b_1_credlmt_228A', 'count_bureau_b_1_credlmt_3940954A', 'count_bureau_b_1_credor_3940957M', 'count_bureau_b_1_credquantity_1099L', 'count_bureau_b_1_credquantity_984L', 'count_bureau_b_1_debtpastduevalue_732A', 'count_bureau_b_1_debtvalue_227A', 'count_bureau_b_1_dpd_550P', 'count_bureau_b_1_dpd_733P', 'count_bureau_b_1_dpdmax_851P', 'count_bureau_b_1_dpdmaxdatemonth_804T', 'count_bureau_b_1_dpdmaxdateyear_742T', 'count_bureau_b_1_installmentamount_644A', 'count_bureau_b_1_installmentamount_833A', 'count_bureau_b_1_instlamount_892A', 'count_bureau_b_1_interesteffectiverate_369L', 'count_bureau_b_1_interestrateyearly_538L', 'count_bureau_b_1_lastupdate_260D', 'count_bureau_b_1_maxdebtpduevalodued_3940955A', 'count_bureau_b_1_num_group1', 'count_bureau_b_1_numberofinstls_810L', 'count_bureau_b_1_overdueamountmax_950A', 'count_bureau_b_1_overdueamountmaxdatemonth_494T', 'count_bureau_b_1_overdueamountmaxdateyear_432T', 'count_bureau_b_1_periodicityofpmts_997L', 'count_bureau_b_1_periodicityofpmts_997M', 'count_bureau_b_1_pmtdaysoverdue_1135P', 'count_bureau_b_1_pmtmethod_731M', 'count_bureau_b_1_pmtnumpending_403L', 'count_bureau_b_1_purposeofcred_722M', 'count_bureau_b_1_residualamount_1093A', 'count_bureau_b_1_residualamount_127A', 'count_bureau_b_1_residualamount_3940956A', 'count_bureau_b_1_subjectrole_326M', 'count_bureau_b_1_subjectrole_43M', 'count_bureau_b_1_totalamount_503A', 'count_bureau_b_1_totalamount_881A', 'count_bureau_b_2_num_group1', 'count_bureau_b_2_num_group2', 'count_bureau_b_2_pmts_date_1107D', 'count_bureau_b_2_pmts_dpdvalue_108P', 'count_bureau_b_2_pmts_pmtsoverdue_635A', 'count_other_amtdebitincoming_4809443A', 'count_other_amtdebitoutgoing_4809440A', 'count_other_amtdepositbalance_4809441A', 'count_other_amtdepositincoming_4809444A', 'count_other_amtdepositoutgoing_4809442A', 'count_other_num_group1', 'count_person1_birth_259D', 'count_person1_incometype_1044T', 'count_person1_mainoccupationinc_384A', 'count_person1_sex_738L', 'count_static_cb_description_5085714M', 'count_static_cb_education_1103M', 'count_static_cb_education_88M', 'count_static_cb_maritalst_385M', 'count_static_cb_maritalst_893M','first_pmts_dpd_303P','first_pmts_overdue_1152A']\n    # 모델 학습 시 1500번의 반복 동안 이 특성들이 5번 미만으로 사용됨.\n    useless_cols=['count_applprev1_credacc_transactions_402L', 'last_applprev2_cacccardblochreas_147M', 'last_credit_bureau_a_1_classificationofcontr_13M', 'last_credit_bureau_a_1_contractst_545M', 'count_credit_bureau_a_1_debtoutstand_525A', 'count_credit_bureau_a_1_debtoverdue_47A', 'last_credit_bureau_a_1_description_351M', 'last_credit_bureau_a_1_financialinstitution_591M', 'count_credit_bureau_a_1_overdueamountmax2_14A', 'last_credit_bureau_a_1_purposeofcred_426M', 'last_credit_bureau_a_1_subjectrole_93M', 'count_credit_bureau_a_1_totalamount_6A', 'max_collater_typofvalofguarant_407M', 'last_collater_typofvalofguarant_407M', 'last_collaterals_typeofguarante_359M', 'last_collaterals_typeofguarante_669M', 'last_subjectroles_name_541M', 'count_person1_birthdate_87D', 'count_person1_contaddr_matchlist_1032L', 'count_person1_contaddr_smempladdr_334L', 'count_person1_contaddr_zipcode_807M', 'count_person1_education_927M', 'max_person1_empladdr_district_926M', 'max_person1_empladdr_zipcode_114M', 'count_person1_empladdr_zipcode_114M', 'count_person1_gender_992L', 'count_person1_housingtype_772L', 'count_person1_isreference_387L', 'max_person1_persontype_1072L', 'max_person1_persontype_792L', 'count_person1_persontype_792L', 'count_person1_registaddr_district_1083M', 'count_person1_role_993L', 'count_person1_safeguarantyflag_411L', 'max_person2_conts_role_79M', 'max_person2_empls_economicalst_849M', 'last_person2_empls_economicalst_849M', 'applicationcnt_361L', 'clientscnt_157L', 'clientscnt_257L', 'count_static_cb_days120_123L', 'count_static_cb_days180_256L', 'count_static_cb_days30_165L', 'count_static_cb_days360_512L', 'count_static_cb_days90_310L', 'last_static_cb_education_88M', 'count_static_cb_firstquarter_103L', 'count_static_cb_formonth_118L', 'count_static_cb_formonth_206L', 'count_static_cb_formonth_535L', 'count_static_cb_forquarter_1017L', 'count_static_cb_forquarter_462L', 'count_static_cb_forquarter_634L', 'count_static_cb_fortoday_1092L', 'count_static_cb_forweek_1077L', 'count_static_cb_forweek_528L', 'count_static_cb_forweek_601L', 'count_static_cb_foryear_618L', 'count_static_cb_foryear_818L', 'count_static_cb_foryear_850L', 'count_static_cb_fourthquarter_440L', 'last_static_cb_maritalst_893M', 'count_static_cb_numberofqueries_373L', 'count_static_cb_secondquarter_766L', 'count_static_cb_thirdquarter_1082L', 'count_applprev1_credacc_minhisbal_90A', 'max_credit_bureau_a_1_overdueamount_31A', 'count_credit_bureau_a_1_totaloutstanddebtvalue_39A', 'count_person1_childnum_185L', 'count_person1_empladdr_district_926M', 'count_person1_language1_981M', 'count_person1_personindex_1023L', 'count_person1_registaddr_zipcode_184M', 'count_person1_remitter_829L', 'count_person1_type_25L', 'last_person2_conts_role_79M', 'clientscnt_100L', 'count_applprev1_credacc_maxhisbal_375A', 'count_applprev1_credacc_status_367L', 'count_applprev2_credacc_cards_status_52L', 'applicationscnt_629L', 'numpmtchanneldd_318L', 'count_credit_bureau_a_1_dateofcredend_289D', 'count_credit_bureau_a_1_dateofcredend_353D', 'count_collater_valueofguarantee_1124L', 'count_person1_contaddr_district_15M', 'count_person1_persontype_1072L', 'applications30d_658L', 'clientscnt3m_3712950L', 'clientscnt_304L', 'clientscnt_493L', 'count_deposit_openingdate_313D', 'clientscnt_1130L', 'count_credit_bureau_a_1_interestrate_508L', 'count_credit_bureau_a_1_overdueamountmax_155A', 'count_credit_bureau_a_1_overdueamountmaxdateyear_2T', 'count_person1_maritalst_703L', 'count_person1_num_group1', 'count_static_cb_for3years_584L', 'count_credit_bureau_a_1_refreshdate_3813885D', 'count_credit_bureau_a_1_totaldebtoverduevalue_178A', 'max_person1_education_927M', 'numactivecreds_622L', 'max_applprev1_actualdpd_943P', 'count_applprev1_credacc_actualbalance_314A', 'count_applprev1_revolvingaccount_394A', 'count_credit_bureau_a_1_dateofcredstart_739D', 'count_credit_bureau_a_1_dpdmax_757P', 'count_credit_bureau_a_1_lastupdate_1112D', 'max_credit_bureau_a_1_numberofoverdueinstls_834L', 'max_credit_bureau_a_1_outstandingamount_354A', 'count_credit_bureau_a_1_overdueamountmaxdatemonth_284T', 'count_credit_bureau_a_1_totaldebtoverduevalue_718A', 'max_collaterals_typeofguarante_359M', 'count_debitcard_last180dayturnover_1134A', 'count_debitcard_openingdate_857D', 'count_person1_relationshiptoclient_415T', 'count_person1_role_1084L', 'applicationscnt_464L', 'numactiverelcontr_750L', 'numnotactivated_1143L', 'count_static_cb_dateofbirth_337D', 'max_static_cb_education_88M', 'max_static_cb_maritalst_893M', 'count_static_cb_pmtcount_4527229L', 'count_credit_bureau_a_1_overdueamountmax_35A']\n    \nimport random # 난수 생성 함수를 제공함\n# 랜덤 시드를 설정하여 모델 결과를 재현할 수 있게 함\ndef seed_everything(seed):\n    np.random.seed(seed) # numpy의 랜덤 시드 설정\n    random.seed(seed) # python의 내장 랜덤 시드 설정\nseed_everything(Config.seed)\nprint(f\"len(Config.drop_cols):{len(Config.drop_cols)},len(Config.useless_cols):{len(Config.useless_cols)}\")\n\n# 학습 데이터에서 각 특성의 데이터 타입(dtype)을 읽음\ncolname2dtype=pd.read_csv(\"/kaggle/input/home-credit-inconsistent-data-types/colname2dtype.csv\")\ncolname=colname2dtype['Column'].values\ndtype=colname2dtype['DataType'].values\n\ndtype2pl={}\ndtype2pl['Int64']=pl.Int64\ndtype2pl['Float64']=pl.Float32\ndtype2pl['String']=pl.String\ndtype2pl['Boolean']=pl.String\n\ncolname2dtype={}\nfor idx in range(len(colname)):\n    colname2dtype[colname[idx]]=dtype2pl[dtype[idx]]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.798192Z","iopub.execute_input":"2024-04-14T00:30:21.798737Z","iopub.status.idle":"2024-04-14T00:30:21.861965Z","shell.execute_reply.started":"2024-04-14T00:30:21.798708Z","shell.execute_reply":"2024-04-14T00:30:21.861025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_agg_feature(feats,df,name='base'):\n    for col in df.drop('case_id').columns:\n        feat=df.group_by('case_id').agg( \n                                        pl.max(col).alias(f\"max_{name}_{col}\"),\n                                        pl.std(col).alias(f\"std_{name}_{col}\"),\n                                        pl.last(col).alias(f\"last_{name}_{col}\"),\n                                        pl.count(col).alias(f\"count_{name}_{col}\"),\n                                        )\n        feats=feats.join(feat,on='case_id',how='left')\n    for col in feats.drop(['case_id']).columns:\n        if (col in Config.drop_cols) or (col in Config.useless_cols):\n            feats=feats.drop([col])\n    return feats","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.863038Z","iopub.execute_input":"2024-04-14T00:30:21.863307Z","iopub.status.idle":"2024-04-14T00:30:21.872892Z","shell.execute_reply.started":"2024-04-14T00:30:21.863285Z","shell.execute_reply":"2024-04-14T00:30:21.872036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#테이블에서 각 case_id 찾기\ndef find_agg_feats_per_caseid(df):\n    max_feats=pl.DataFrame({\"case_id\":df['case_id'].unique()})\n    for col in df.drop(['case_id']).columns:\n        feat=df.group_by('case_id').agg(  pl.max(col).alias(f\"max_{col}\"),\n                                          pl.mean(col).alias(f\"mean_{col}\"),\n                                          pl.std(col).alias(f\"std_{col}\"),\n                                          pl.first(col).alias(f\"first_{col}\"),\n                                          pl.last(col).alias(f\"last_{col}\"),\n                                          pl.count(col).alias(f\"count_{col}\"),\n                                          pl.n_unique(col).alias(f\"nunique_{col}\"),\n                                       )\n        max_feats=max_feats.join(feat,on='case_id',how='left')\n    for col in max_feats.drop(['case_id']).columns:\n        if (col in Config.drop_cols) or (col in Config.useless_cols):\n            max_feats=max_feats.drop([col])\n    return max_feats","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.874810Z","iopub.execute_input":"2024-04-14T00:30:21.875093Z","iopub.status.idle":"2024-04-14T00:30:21.883500Z","shell.execute_reply.started":"2024-04-14T00:30:21.875070Z","shell.execute_reply":"2024-04-14T00:30:21.882679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df):\n    for col in df.columns:\n        df=df.with_columns(pl.col(col).cast(colname2dtype[col]).alias(col))\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.884433Z","iopub.execute_input":"2024-04-14T00:30:21.884690Z","iopub.status.idle":"2024-04-14T00:30:21.894510Z","shell.execute_reply.started":"2024-04-14T00:30:21.884666Z","shell.execute_reply":"2024-04-14T00:30:21.893862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#테이블 df의 모든 열을 순회하고 데이터 유형을 수정하여 메모리 사용량을 줄입니다.\ndef reduce_mem_usage(df, float16_as32=True):\n    #memory_usage()는 df의 각 열의 메모리 사용량이고 sum은 그 합계입니다. B->KB->MB\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns: # 데이터프레임의 모든 열을 순회\n        col_type = df[col].dtype # 해당 열의 데이터 타입을 가져옴\n        if col_type != object and str(col_type) != 'category': # 데이터 타입이 object나 category가 아닌 경우, 즉 수치형 변수를 처리\n            c_min, c_max = df[col].min(), df[col].max() # 해당 열의 최소값과 최대값을 계산\n            if str(col_type)[:3] == 'int': # 열의 타입이 정수형(int)인 경우\n                # 해당 열의 데이터가 int8 범위에 들어가는지 확인하고 타입을 변환 (-128에서 127까지)\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                # 해당 열의 데이터가 int16 범위에 들어가는지 확인하고 타입을 변환 (-32,768에서 32,767까지)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                # 해당 열의 데이터가 int32 범위에 들어가는지 확인하고 타입을 변환 (-2,147,483,648에서 2,147,483,647까지)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                # 해당 열의 데이터가 int64 범위에 들어가는지 확인하고 타입을 변환 (-9,223,372,036,854,775,808에서 9,223,372,036,854,775,807까지)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else: # 열의 타입이 실수형(float)인 경우\n                # 해당 열의 데이터가 float16 범위에 들어가는지 확인하고 타입을 변환, 더 높은 정밀도가 필요한 경우 float32 사용을 고려\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32: # 더 높은 정밀도가 필요하면 float32를 선택\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)\n                # 해당 열의 데이터가 float32 범위에 들어가는지 확인하고 타입을 변환\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                # 해당 열의 데이터가 float64 범위에 들어가는지 확인하고 타입을 변환\n                else:\n                    df[col] = df[col].astype(np.float64)\n    \n    # 최적화 후의 메모리 사용량 계산\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    # 최적화 전 대비 메모리 절감 비율\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.895836Z","iopub.execute_input":"2024-04-14T00:30:21.896109Z","iopub.status.idle":"2024-04-14T00:30:21.909812Z","shell.execute_reply.started":"2024-04-14T00:30:21.896087Z","shell.execute_reply":"2024-04-14T00:30:21.908912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#after break는 파일의 각 특성의 의미를 자세히 연구한 후입니다.\ndef preprocessor(mode='train'):#mode='train'|'test'\n    \n    print(f\"{mode} base file. number:1\")\n    feats=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_base.csv\").pipe(set_table_dtypes)\n    feats=feats.drop(['MONTH'])#'date_decision',\n    \n    print(f\"{mode} applprev_1 file. number:2(3)\")#훈련 데이터는 2개 파일, 테스트 데이터는 3개 파일.\n    applprev1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_applprev_1_0.csv\").pipe(set_table_dtypes)\n    file_num=1+int(mode=='test')#테스트 데이터에는 2개 파일이 더 있고, 훈련 데이터에는 1개 파일이 더 있음,\n    for i in range(file_num):\n        applprev1_i=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_applprev_1_{i+1}.csv\").pipe(set_table_dtypes)\n        applprev1=pl.concat([applprev1,applprev1_i],how=\"vertical_relaxed\")#수직 병합, 데이터 타입 일치 요구를 완화함\n        del applprev1_i\n    feats=find_agg_feature(feats,applprev1,'applprev1')\n    del applprev1\n    gc.collect()#수동으로 가비지 컬렉션을 트리거하여, 가비지 컬렉터가 사용하지 않는 것으로 표시한 메모리를 강제로 회수함\n    \n    print(f\"{mode} applprev_2 file. number:1\")\n    applprev2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_applprev_2.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,applprev2,name='applprev2')\n    del applprev2\n    gc.collect()#수동으로 가비지 컬렉션을 트리거하여, 가비지 컬렉터가 사용하지 않는 것으로 표시한 메모리를 강제로 회수함\n    \n    print(f\"{mode} credit_bureau_a_1 file. number:4(5)\")#훈련 데이터에는 4개 파일, 테스트 데이터에는 5개 파일.\n    credit_bureau_a_1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_1_0.csv\").pipe(set_table_dtypes)\n    file_num=3+int(mode=='test')#测试数据还有4个文件,训练数据还有3个文件,\n    for i in range(file_num):\n        credit_bureau_a_1_i=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_1_{i+1}.csv\").pipe(set_table_dtypes)\n        credit_bureau_a_1=pl.concat([credit_bureau_a_1,credit_bureau_a_1_i],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n        del credit_bureau_a_1_i\n        gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    feats=find_agg_feature(feats,credit_bureau_a_1,'credit_bureau_a_1')\n    del credit_bureau_a_1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} credit_bureau_a_2 file. number:11(12)\")#训练数据11个文件,测试数据12个文件.\n    credit_bureau_a_2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_2_0.csv\").pipe(set_table_dtypes)\n    credit_bureau_a_2_max=find_agg_feats_per_caseid(credit_bureau_a_2)\n    file_num=10+int(mode=='test')#测试数据还有10个文件,训练数据还有11个文件,\n    for i in range(file_num):\n        credit_bureau_a_2_i=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_a_2_{i+1}.csv\").pipe(set_table_dtypes)\n        credit_bureau_a_2_i_max=find_agg_feats_per_caseid(credit_bureau_a_2_i)\n        credit_bureau_a_2_max=pl.concat([credit_bureau_a_2_max,credit_bureau_a_2_i_max],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n        del credit_bureau_a_2_i,credit_bureau_a_2_i_max\n    feats=feats.join(credit_bureau_a_2_max,on='case_id',how='left')\n    del credit_bureau_a_2_max\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} credit_bureau_b file. number:2\")\n    bureau_b_1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_b_1.csv\").pipe(set_table_dtypes)\n    bureau_b_2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_credit_bureau_b_2.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,bureau_b_1,name='bureau_b_1')\n    feats=find_agg_feature(feats,bureau_b_2,name='bureau_b_2')\n    del bureau_b_1,bureau_b_2\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n\n    print(f\"{mode} debitcard file. number:1\")\n    debitcard=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_debitcard_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,debitcard,name='debitcard')\n    del debitcard\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} deposit file. number:1\")\n    deposit=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_deposit_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,deposit,name='deposit')\n    del deposit\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} other file. number:1\")\n    other=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_other_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,other,name='other')\n    del other\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} person1 file. number:1\")\n    person1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_person_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,person1,name='person1')   \n    del person1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n\n    print(f\"{mode} person2 file. number:1\")\n    #经过检查person2训练集和测试集对应的列dtype都对应的上\n    person2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_person_2.csv\").pipe(set_table_dtypes)    \n    feats=find_agg_feature(feats,person2,name='person2')\n    del person2\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} static_0 file. number:2(3)\")\n    #pipe用于在DataFrame上自定义自己的函数\n    static_0_0=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_0_0.csv\").pipe(set_table_dtypes)\n    static_0_1=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_0_1.csv\").pipe(set_table_dtypes)\n    static=pl.concat([static_0_0,static_0_1],how=\"vertical_relaxed\")#垂直合并,并且放宽了数据类型匹配的限制\n    if mode=='test':#如果是测试数据的话还有一个文件\n        static_0_2=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_0_2.csv\").pipe(set_table_dtypes)\n        static=pl.concat([static,static_0_2],how=\"vertical_relaxed\")\n    feats=feats.join(static,on='case_id',how='left')\n    del static,static_0_0,static_0_1\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} static_cb_file. number:1\")\n    static_cb=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_cb_0.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,static_cb,name='static_cb')\n    del static_cb\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} tax_a file. number:1\")\n    tax_a=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_tax_registry_a_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,tax_a,name='tax_a')\n    del tax_a\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n   \n    print(f\"{mode} tax_b file. number:1\")\n    tax_b=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_tax_registry_b_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,tax_b,name='tax_b')\n    del tax_b\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    \n    print(f\"{mode} tax_c file. number:1\")\n    tax_c=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_tax_registry_c_1.csv\").pipe(set_table_dtypes)\n    if len(tax_c)==0:\n        tax_c=pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_tax_registry_c_1.csv\").pipe(set_table_dtypes)\n    feats=find_agg_feature(feats,tax_c,name='tax_c')\n    del tax_c\n    gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n    print(\"-\"*30)\n    \n    return feats","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.911185Z","iopub.execute_input":"2024-04-14T00:30:21.911450Z","iopub.status.idle":"2024-04-14T00:30:21.938403Z","shell.execute_reply.started":"2024-04-14T00:30:21.911428Z","shell.execute_reply":"2024-04-14T00:30:21.937538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats=preprocessor(mode='train')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:30:21.939367Z","iopub.execute_input":"2024-04-14T00:30:21.939618Z","iopub.status.idle":"2024-04-14T00:41:24.334304Z","shell.execute_reply.started":"2024-04-14T00:30:21.939591Z","shell.execute_reply":"2024-04-14T00:41:24.333362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats=train_feats.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:41:24.335460Z","iopub.execute_input":"2024-04-14T00:41:24.335771Z","iopub.status.idle":"2024-04-14T00:41:35.826323Z","shell.execute_reply.started":"2024-04-14T00:41:24.335745Z","shell.execute_reply":"2024-04-14T00:41:35.825124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats=reduce_mem_usage(train_feats, float16_as32=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:41:35.831435Z","iopub.execute_input":"2024-04-14T00:41:35.831761Z","iopub.status.idle":"2024-04-14T00:41:48.094864Z","shell.execute_reply.started":"2024-04-14T00:41:35.831735Z","shell.execute_reply":"2024-04-14T00:41:48.093800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:41:48.095930Z","iopub.execute_input":"2024-04-14T00:41:48.096223Z","iopub.status.idle":"2024-04-14T00:41:48.135820Z","shell.execute_reply.started":"2024-04-14T00:41:48.096198Z","shell.execute_reply":"2024-04-14T00:41:48.134868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_feats=preprocessor(mode='test')\ntest_feats=test_feats.to_pandas()\ntest_feats=reduce_mem_usage(test_feats, float16_as32=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:41:48.136829Z","iopub.execute_input":"2024-04-14T00:41:48.137107Z","iopub.status.idle":"2024-04-14T00:41:53.122980Z","shell.execute_reply.started":"2024-04-14T00:41:48.137076Z","shell.execute_reply":"2024-04-14T00:41:53.121921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_feats.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:41:53.124157Z","iopub.execute_input":"2024-04-14T00:41:53.124441Z","iopub.status.idle":"2024-04-14T00:41:53.157445Z","shell.execute_reply.started":"2024-04-14T00:41:53.124416Z","shell.execute_reply":"2024-04-14T00:41:53.156567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_cols=[]\nfor col in test_feats:\n    if (col[-1]=='D') and 'count' not in col:\n        date_cols.append(col)\nprint(f\"date_cols:{date_cols}\")\n\n#对日期列变量的处理\ndef deal_date(df):\n    df['date_decision']=pd.to_datetime(df['date_decision'])\n    for col in date_cols:\n        print(f\"col:{col}\")\n        df[col]=pd.to_datetime(df[col])\n        df[f\"{col}_date_decision_gap_day\"]=(df[col]-df['date_decision']).dt.total_seconds() // 86400        \n        df.drop([col],axis=1,inplace=True)\n    df['month_decision'] = df[\"date_decision\"].dt.month\n    df['weekday_decision'] = df[\"date_decision\"].dt.weekday\n    df.drop(['date_decision'],axis=1,inplace=True)\n    return df\ntrain_feats=deal_date(train_feats)\ntest_feats=deal_date(test_feats)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:41:53.158492Z","iopub.execute_input":"2024-04-14T00:41:53.158781Z","iopub.status.idle":"2024-04-14T00:47:02.196376Z","shell.execute_reply.started":"2024-04-14T00:41:53.158757Z","shell.execute_reply":"2024-04-14T00:47:02.195471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#对字符串特征列进行独热编码的转换\nprint(\"----------string one hot encoder ****\")\n            \nfor col in test_feats.columns:\n    if col!='WEEK_NUM':\n        n_unique=train_feats[col].nunique()\n        if n_unique==2 and train_feats[col].dtype=='object':\n            print(f\"one_hot_2:{col}\")\n            unique=train_feats[col].unique()\n            #随便选择一个类别进行转换,比如gender='Female'\n            train_feats[col]=(train_feats[col]==unique[0]).astype(int)\n            test_feats[col]=(test_feats[col]==unique[0]).astype(int)\n        elif n_unique<50 and train_feats[col].dtype=='object':#如果是类别型变量\n            train_feats[col]=(train_feats[col]).astype(\"category\")\n            test_feats[col]=(test_feats[col]).astype(\"category\")\n\nprint(\"----------drop other string or unique value or full null value ****\")\ndrop_cols=[]\nfor col in test_feats.columns:\n    if (train_feats[col].dtype=='object') or (train_feats[col].nunique()==1) or train_feats[col].isna().mean()>0.95:\n        drop_cols+=[col]\nprint(f\"len(drop_cols):{len(drop_cols)},drop_cols:{drop_cols}\")\n#case_id的话就当作普通的id,WEEK_NUM测试数据比训练数据大.\ndrop_cols+=['case_id','WEEK_NUM']\ntrain_feats.drop(drop_cols,axis=1,inplace=True)\ntest_feats.drop(drop_cols,axis=1,inplace=True)\n\ntrain_feats=reduce_mem_usage(train_feats, float16_as32=False)\ntest_feats=reduce_mem_usage(test_feats, float16_as32=False)\n\nprint(f\"len(train_feats):{len(train_feats)},total_features_counts:{len(test_feats.columns)}\")\ntrain_feats.head()\n\n# #数值类型的变量和类别型变量\n# num_cols=[]\n# cat_cols=[]\n# for col in test_feats.columns:\n#     if str(train_feats[col].dtype)!='category':\n#         num_cols.append(col)\n#     else:\n#         cat_cols.append(col)\n# print(f\"len(num_cols):{len(num_cols)},num_cols:{num_cols}\")\n# #根据nan值对col进行分组\n# groupby_nancnt={}\n# for col in num_cols:\n#     nancnt=train_feats[col].isna().sum()#这列缺失值有多少个\n#     #如果是第一个创建列表[col],如果有新的col继续加\n#     try:\n#         groupby_nancnt[nancnt]=[col]\n#     except:\n#         groupby_nancnt[nancnt].append(col)  \n# print(f\"groupby_nancnt:{groupby_nancnt}\")\n\n# def pearson_corr(x1,x2):\n#     \"\"\"\n#     x1,x2:np.array\n#     \"\"\"\n#     mean_x1=np.mean(x1)\n#     mean_x2=np.mean(x2)\n#     std_x1=np.std(x1)\n#     std_x2=np.std(x2)\n#     pearson=np.mean((x1-mean_x1)*(x2-mean_x2))/(std_x1*std_x2)\n#     return pearson\n# #选择特征\n# choose_cols=[]\n# for key,value in groupby_nancnt.items():\n#     if len(value)==1:#如果就是一个特征,那就直接保留就行了\n#         choose_cols+=value\n#     else:#如果缺失值数量=key的列数>1,在相关性大的几列中取nunique最多的一列\n#         remain_cols=value\n#         is_choose=np.zeros(len(remain_cols))#每列是否被选中\n#         for i in range(len(remain_cols)):\n#             groups=[remain_cols[i]]\n#             for j in range(i+1,len(remain_cols)):\n#                 tmp_df=train_feats[remain_cols[i],remain_cols[j]].copy().dropna()\n#                 if abs(pearson_corr(tmp_df[0].values,tmp_df[1].values))>0.8:\n#                     if is_choose[j]==0:\n#                         groups.append(remain_cols[j])\n#                         is_choose[j]=1\n#             max_idx=0;max_unique=1\n#             for idx in range(len(groups)):\n#                 cur_unique=train_feats[groups[idx]].nunique()\n#                 if cur_unique>max_unique():\n#                     max_unique=cur_unique\n#                     max_idx=idx\n#             choose_cols.append(groups[max_idx])\n# choose_cols+=cat_cols\n# print(f\"origin_features_counts:{len(test_feats.columns)},now_features_counts:{len(choose_cols)}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:47:02.197740Z","iopub.execute_input":"2024-04-14T00:47:02.198038Z","iopub.status.idle":"2024-04-14T00:48:10.367593Z","shell.execute_reply.started":"2024-04-14T00:47:02.198013Z","shell.execute_reply":"2024-04-14T00:48:10.366679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#0405 mean_gini:0.7230855268034984 \nchoose_cols=[col for col in test_feats.columns]\n#保存训练好的树模型,obj是保存的模型,path是需要保存的路径","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.369055Z","iopub.execute_input":"2024-04-14T00:48:10.369512Z","iopub.status.idle":"2024-04-14T00:48:10.374341Z","shell.execute_reply.started":"2024-04-14T00:48:10.369478Z","shell.execute_reply":"2024-04-14T00:48:10.373340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(choose_cols)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.375428Z","iopub.execute_input":"2024-04-14T00:48:10.375723Z","iopub.status.idle":"2024-04-14T00:48:10.388683Z","shell.execute_reply.started":"2024-04-14T00:48:10.375699Z","shell.execute_reply":"2024-04-14T00:48:10.387921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Importance Top 10\n1.\nlastapprcommoditycat_1041M\t\nCommodity category of the last loan applications made by the applicant.\t\n신청자가 마지막으로 대출 신청한 상품 카테고리입니다\n\n2.\nlastrejectcommoditycat_161M\t\nCategory of commodity in the applicant's last rejected application.\t\n신청자의 마지막으로 거부된 신청에서의 상품 카테고리\n\n3.\nrelationshiptoclient_642T\t\nRelationship to the client.\t\n클라이언트와의 관계.\t\n개인이나 조직이 클라이언트와 어떤 관계를 가지고 있는지를 나타냅니다. \n예를 들어, 고객, 파트너, 공급업체, 직원 등 다양한 관계 유형이 있을 수 있습니다.\n\n4.\nincometype_1044T\t\nType of income of the person\t\n개인의 소득 유형\n\n5.\nprice_1097A\t\nCredit price.\t\n신용 가격.\n신용 상품, 서비스에 대한 가격이나 비용\n대출의 이자율, 신용카드의 연회비, 신용 기반의 다른 서비스의 수수료\n\n6.\nresidualamount_856A\t\nResidual amount for the active contract.\t\n활성 계약의 잔여 금액(아직 지불되지 않은 금액)\n원금, 이자 등 납부 해야하는 금액\n\n7.\nrelationshiptoclient_415T\t\nRelationship to the client.\n클라이언트와의 관계.\n\n8.\noverdueamountmax2date_1142D\t\nDate of maximal past due amount for an active contract.\t\n활성 계약에서 발생한 최대 연체 금액의 발생 날짜\n\n9.\npmtnum_254L\t\nTotal number of loan payments made by the client.\t\n대출을 받은 클라이언트가 현재까지 상환한 대출금의 총 회차 수\n\n10.\ninterestrate_311L\t\nThe interest rate of the active credit contract.\t\n활성 신용 계약의 이자율입니다.","metadata":{}},{"cell_type":"code","source":"'''\nlastapprcommoditycat_1041M\nlastrejectcommoditycat_161M\nlast_person1_relationshiptoclient_642T\nmax_person1_incometype_1044T\nprice_1097A\nmax_credit_bureau_a_1_residualamount_856A\nmax_person1_relationshiptoclient_415T\nmax_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day\npmtnum_254L\ninterestrate_311L\n'''\ntrain_feats.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.389778Z","iopub.execute_input":"2024-04-14T00:48:10.390061Z","iopub.status.idle":"2024-04-14T00:48:10.424655Z","shell.execute_reply.started":"2024-04-14T00:48:10.390037Z","shell.execute_reply":"2024-04-14T00:48:10.423806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# max_person1_relationshiptoclient_642T 와 max_person1_relationshiptoclient_415T 는 같다.","metadata":{}},{"cell_type":"code","source":"train_feats['max_person1_relationshiptoclient_642T'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.425845Z","iopub.execute_input":"2024-04-14T00:48:10.426476Z","iopub.status.idle":"2024-04-14T00:48:10.445861Z","shell.execute_reply.started":"2024-04-14T00:48:10.426438Z","shell.execute_reply":"2024-04-14T00:48:10.444940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['max_person1_relationshiptoclient_415T'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.447135Z","iopub.execute_input":"2024-04-14T00:48:10.447443Z","iopub.status.idle":"2024-04-14T00:48:10.465536Z","shell.execute_reply.started":"2024-04-14T00:48:10.447418Z","shell.execute_reply":"2024-04-14T00:48:10.464681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['last_person1_relationshiptoclient_642T'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:51:16.431427Z","iopub.execute_input":"2024-04-14T00:51:16.432217Z","iopub.status.idle":"2024-04-14T00:51:16.445273Z","shell.execute_reply.started":"2024-04-14T00:51:16.432180Z","shell.execute_reply":"2024-04-14T00:51:16.444117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# max_credit_bureau_a_1_residualamount_856A\nResidual amount for the active contract  \n현재 활성화된 계약의 잔여 금액\n","metadata":{}},{"cell_type":"code","source":"train_feats['max_credit_bureau_a_1_residualamount_856A'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.466798Z","iopub.execute_input":"2024-04-14T00:48:10.467069Z","iopub.status.idle":"2024-04-14T00:48:10.602150Z","shell.execute_reply.started":"2024-04-14T00:48:10.467046Z","shell.execute_reply":"2024-04-14T00:48:10.601177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['max_credit_bureau_a_1_residualamount_856A'].value_counts(dropna=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.603214Z","iopub.execute_input":"2024-04-14T00:48:10.603468Z","iopub.status.idle":"2024-04-14T00:48:10.703745Z","shell.execute_reply.started":"2024-04-14T00:48:10.603445Z","shell.execute_reply":"2024-04-14T00:48:10.702783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_0_df = train_feats[train_feats['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:10.704913Z","iopub.execute_input":"2024-04-14T00:48:10.705194Z","iopub.status.idle":"2024-04-14T00:48:13.791767Z","shell.execute_reply.started":"2024-04-14T00:48:10.705172Z","shell.execute_reply":"2024-04-14T00:48:13.790728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['max_credit_bureau_a_1_residualamount_856A']","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.793141Z","iopub.execute_input":"2024-04-14T00:48:13.793816Z","iopub.status.idle":"2024-04-14T00:48:13.803139Z","shell.execute_reply.started":"2024-04-14T00:48:13.793780Z","shell.execute_reply":"2024-04-14T00:48:13.802007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['max_credit_bureau_a_1_residualamount_856A']","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.804442Z","iopub.execute_input":"2024-04-14T00:48:13.805335Z","iopub.status.idle":"2024-04-14T00:48:13.816202Z","shell.execute_reply.started":"2024-04-14T00:48:13.805301Z","shell.execute_reply":"2024-04-14T00:48:13.815274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['max_credit_bureau_a_1_residualamount_856A'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.817680Z","iopub.execute_input":"2024-04-14T00:48:13.817959Z","iopub.status.idle":"2024-04-14T00:48:13.833379Z","shell.execute_reply.started":"2024-04-14T00:48:13.817936Z","shell.execute_reply":"2024-04-14T00:48:13.832283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['max_credit_bureau_a_1_residualamount_856A'].value_counts(dropna=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.834525Z","iopub.execute_input":"2024-04-14T00:48:13.835195Z","iopub.status.idle":"2024-04-14T00:48:13.848128Z","shell.execute_reply.started":"2024-04-14T00:48:13.835169Z","shell.execute_reply":"2024-04-14T00:48:13.847266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_count = target_1_df['max_credit_bureau_a_1_residualamount_856A'].isnull().sum()\n\n# 전체 데이터 개수 계산\ntotal_count = len(target_1_df)\n\n# 널값의 비율 계산\nnull_percentage = (null_count / total_count) * 100\n\n# 결과 출력\nprint(\"널값 비율: {:.2f}%\".format(null_percentage))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.855733Z","iopub.execute_input":"2024-04-14T00:48:13.856045Z","iopub.status.idle":"2024-04-14T00:48:13.862511Z","shell.execute_reply.started":"2024-04-14T00:48:13.856019Z","shell.execute_reply":"2024-04-14T00:48:13.861411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_count = target_0_df['max_credit_bureau_a_1_residualamount_856A'].isnull().sum()\n\n# 전체 데이터 개수 계산\ntotal_count = len(target_0_df)\n\n# 널값의 비율 계산\nnull_percentage = (null_count / total_count) * 100\n\n# 결과 출력\nprint(\"널값 비율: {:.2f}%\".format(null_percentage))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.863756Z","iopub.execute_input":"2024-04-14T00:48:13.864077Z","iopub.status.idle":"2024-04-14T00:48:13.873503Z","shell.execute_reply.started":"2024-04-14T00:48:13.864048Z","shell.execute_reply":"2024-04-14T00:48:13.872510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 데이터가 불균형하므로 common_norm=False로 설정합니다.\nplt.figure(figsize=(10, 6))\n\nsns.kdeplot(\n    train_feats[train_feats['target'] == 1]['max_credit_bureau_a_1_residualamount_856A'],\n    color='orange',\n    label='target1 (n=20089)',\n    common_norm=False,  # 각 그룹의 밀도를 별도로 정규화합니다.\n    fill=True  # 밀도 아래의 공간을 색으로 채웁니다.\n)\n\nsns.kdeplot(\n    train_feats[train_feats['target'] == 0]['max_credit_bureau_a_1_residualamount_856A'],\n    color='blue',\n    label='target0 (n=570354)',\n    common_norm=False,\n    fill=True\n)\n\nplt.xlabel('max_credit_bureau_a_1_residualamount_856A')\nplt.ylabel('Density')\nplt.title('Normalized KDE of max_credit_bureau_a_1_residualamount_856A for target1 and target0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:13.874749Z","iopub.execute_input":"2024-04-14T00:48:13.875384Z","iopub.status.idle":"2024-04-14T00:48:22.443546Z","shell.execute_reply.started":"2024-04-14T00:48:13.875350Z","shell.execute_reply":"2024-04-14T00:48:22.442575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nplt.figure(figsize=(10, 6))\n\n# 로그 스케일을 적용하여 그래프 그리기\nsns.kdeplot(\n    np.log1p(train_feats[train_feats['target'] == 1]['max_credit_bureau_a_1_residualamount_856A']),\n    color='orange',\n    label='target1 (n=20089)',\n    fill=True\n)\n\nsns.kdeplot(\n    np.log1p(train_feats[train_feats['target'] == 0]['max_credit_bureau_a_1_residualamount_856A']),\n    color='blue',\n    label='target0 (n=570354)',\n    fill=True\n)\n\nplt.xlabel('Log of max_credit_bureau_a_1_residualamount_856A')\nplt.ylabel('Density')\nplt.title('Log Scale KDE of max_credit_bureau_a_1_residualamount_856A for target1 and target0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:22.445243Z","iopub.execute_input":"2024-04-14T00:48:22.445976Z","iopub.status.idle":"2024-04-14T00:48:30.491257Z","shell.execute_reply.started":"2024-04-14T00:48:22.445936Z","shell.execute_reply":"2024-04-14T00:48:30.490222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 특정 백분위수를 기준으로 최대값을 정합니다. 예를 들어, 95% 백분위수를 최대값으로 설정합니다.\nupper_bound = np.percentile(\n    train_feats['max_credit_bureau_a_1_residualamount_856A'].dropna(), 95)\n\nplt.figure(figsize=(10, 6))\n\n# x축 범위를 제한하여 그래프 그리기\nsns.kdeplot(\n    train_feats[train_feats['target'] == 1]['max_credit_bureau_a_1_residualamount_856A'].clip(upper=upper_bound),\n    color='orange',\n    label='target1 (n=20089)',\n    fill=True\n)\n\nsns.kdeplot(\n    train_feats[train_feats['target'] == 0]['max_credit_bureau_a_1_residualamount_856A'].clip(upper=upper_bound),\n    color='blue',\n    label='target0 (n=570354)',\n    fill=True\n)\n\nplt.xlabel('max_credit_bureau_a_1_residualamount_856A (clipped at 95th percentile)')\nplt.ylabel('Density')\nplt.title('Clipped KDE of max_credit_bureau_a_1_residualamount_856A for target1 and target0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:30.492779Z","iopub.execute_input":"2024-04-14T00:48:30.493602Z","iopub.status.idle":"2024-04-14T00:48:39.210563Z","shell.execute_reply.started":"2024-04-14T00:48:30.493563Z","shell.execute_reply":"2024-04-14T00:48:39.209540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"두 그래프 모두 target0 (대출금을 갚은 사람)과 target1 (대출금을 갚지 못한 사람) 간에 'max_credit_bureau_a_1_residualamount_856A' 값의 분포에 차이를 보여줍니다.\n\n첫 번째 그래프(로그 스케일):\n- Target0의 분포는 상대적으로 넓게 퍼져 있으며, 더 낮은 값에서 높은 빈도를 보입니다.\n- Target1의 분포는 훨씬 좁고, 특정 값에서 피크(가장 높은 지점)가 나타나는 것으로 보입니다. 이는 target1 그룹이 더 작은 범위의 값에 집중되어 있음을 의미합니다.\n- Target1의 피크가 더 높은 것은 해당 값을 가진 데이터 포인트의 비율이 target0에 비해 상대적으로 많다는 것을 나타냅니다.\n\n두 번째 그래프(95% 백분위수로 클리핑):\n- 대부분의 데이터 포인트가 낮은 값 범위에 집중되어 있는 것으로 보입니다.\n- 두 그룹 모두 낮은 값에서 높은 밀도를 보이며, 특히 target1 그룹이 더 높은 밀도를 보이는 구간이 있습니다. 이는 해당 값이 target1 그룹에서 더 흔하다는 것을 나타낼 수 있습니다.\n- 두 그룹의 분포가 중간 값 범위에서 겹치는 것으로 보이나, target1이 target0보다 더 낮은 값을 가지는 경향이 있는 것으로 보입니다.\n\n이러한 인사이트는 대출금을 갚지 못하는 사람들이 특정 residualamount 값 범위에 더 많이 분포한다는 것을 나타내며, 금융 기관이 이러한 패턴을 분석하여 리스크를 관리하는 데 사용할 수 있습니다. 예를 들어, 대출 승인 과정에서 이러한 residualamount 값의 범위를 고려하여 대출의 위험도를 평가하거나, 대출 조건을 결정하는 데 사용될 수 있습니다.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day","metadata":{}},{"cell_type":"code","source":"train_feats['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day']\ntrain_feats['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].value_counts(dropna=False)\ntrain_feats['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:05:27.798462Z","iopub.execute_input":"2024-04-14T04:05:27.798824Z","iopub.status.idle":"2024-04-14T04:05:27.861520Z","shell.execute_reply.started":"2024-04-14T04:05:27.798794Z","shell.execute_reply":"2024-04-14T04:05:27.860596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T01:39:16.413592Z","iopub.execute_input":"2024-04-14T01:39:16.414259Z","iopub.status.idle":"2024-04-14T01:39:16.436890Z","shell.execute_reply.started":"2024-04-14T01:39:16.414232Z","shell.execute_reply":"2024-04-14T01:39:16.436031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:39.296399Z","iopub.execute_input":"2024-04-14T00:48:39.297055Z","iopub.status.idle":"2024-04-14T00:48:39.333651Z","shell.execute_reply.started":"2024-04-14T00:48:39.297022Z","shell.execute_reply":"2024-04-14T00:48:39.332670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_count = target_1_df['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].isnull().sum()\n\n# 전체 데이터 개수 계산\ntotal_count = len(target_1_df)\n\n# 널값의 비율 계산\nnull_percentage = (null_count / total_count) * 100\n\n# 결과 출력\nprint(\"널값 비율: {:.2f}%\".format(null_percentage))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:39.334827Z","iopub.execute_input":"2024-04-14T00:48:39.335108Z","iopub.status.idle":"2024-04-14T00:48:39.341626Z","shell.execute_reply.started":"2024-04-14T00:48:39.335085Z","shell.execute_reply":"2024-04-14T00:48:39.340581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_count = target_0_df['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].isnull().sum()\n\n# 전체 데이터 개수 계산\ntotal_count = len(target_0_df)\n\n# 널값의 비율 계산\nnull_percentage = (null_count / total_count) * 100\n\n# 결과 출력\nprint(\"널값 비율: {:.2f}%\".format(null_percentage))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:39.342814Z","iopub.execute_input":"2024-04-14T00:48:39.343154Z","iopub.status.idle":"2024-04-14T00:48:39.360875Z","shell.execute_reply.started":"2024-04-14T00:48:39.343113Z","shell.execute_reply":"2024-04-14T00:48:39.360060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 데이터가 불균형하므로 common_norm=False로 설정합니다.\nplt.figure(figsize=(10, 6))\n\nsns.kdeplot(\n    train_feats[train_feats['target'] == 1]['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'],\n    color='orange',\n    label='target1 (n=20089)',\n    common_norm=False,  # 각 그룹의 밀도를 별도로 정규화합니다.\n    fill=True  # 밀도 아래의 공간을 색으로 채웁니다.\n)\n\nsns.kdeplot(\n    train_feats[train_feats['target'] == 0]['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'],\n    color='blue',\n    label='target0 (n=570354)',\n    common_norm=False,\n    fill=True\n)\n\nplt.xlabel('max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day')\nplt.ylabel('Density')\nplt.title('Normalized KDE of max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day for target1 and target0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:39.363061Z","iopub.execute_input":"2024-04-14T00:48:39.363338Z","iopub.status.idle":"2024-04-14T00:48:44.960172Z","shell.execute_reply.started":"2024-04-14T00:48:39.363314Z","shell.execute_reply":"2024-04-14T00:48:44.959172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nplt.figure(figsize=(10, 6))\n\n# 로그 스케일을 적용하여 그래프 그리기\nsns.kdeplot(\n    np.log1p(train_feats[train_feats['target'] == 1]['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day']),\n    color='orange',\n    label='target1 (n=20089)',\n    fill=True\n)\n\nsns.kdeplot(\n    np.log1p(train_feats[train_feats['target'] == 0]['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day']),\n    color='blue',\n    label='target0 (n=570354)',\n    fill=True\n)\n\nplt.xlabel('Log of max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day')\nplt.ylabel('Density')\nplt.title('Log Scale KDE of max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day for target1 and target0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:44.961380Z","iopub.execute_input":"2024-04-14T00:48:44.961709Z","iopub.status.idle":"2024-04-14T00:48:50.535584Z","shell.execute_reply.started":"2024-04-14T00:48:44.961682Z","shell.execute_reply":"2024-04-14T00:48:50.534539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 특정 백분위수를 기준으로 최대값을 정합니다. 예를 들어, 95% 백분위수를 최대값으로 설정합니다.\nupper_bound = np.percentile(\n    train_feats['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].dropna(), 95)\n\nplt.figure(figsize=(10, 6))\n\n# x축 범위를 제한하여 그래프 그리기\nsns.kdeplot(\n    train_feats[train_feats['target'] == 1]['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].clip(upper=upper_bound),\n    color='orange',\n    label='target1 (n=20089)',\n    fill=True\n)\n\nsns.kdeplot(\n    train_feats[train_feats['target'] == 0]['max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day'].clip(upper=upper_bound),\n    color='blue',\n    label='target0 (n=570354)',\n    fill=True\n)\n\nplt.xlabel('max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day (clipped at 95th percentile)')\nplt.ylabel('Density')\nplt.title('Clipped KDE of max_credit_bureau_a_1_overdueamountmax2date_1142D_date_decision_gap_day for target1 and target0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:50.536734Z","iopub.execute_input":"2024-04-14T00:48:50.537016Z","iopub.status.idle":"2024-04-14T00:48:56.225746Z","shell.execute_reply.started":"2024-04-14T00:48:50.536993Z","shell.execute_reply":"2024-04-14T00:48:56.224810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lastapprcommoditycat_1041M","metadata":{}},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_0_df = train_feats[train_feats['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:56.226866Z","iopub.execute_input":"2024-04-14T00:48:56.227162Z","iopub.status.idle":"2024-04-14T00:48:59.592514Z","shell.execute_reply.started":"2024-04-14T00:48:56.227137Z","shell.execute_reply":"2024-04-14T00:48:59.591455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['lastapprcommoditycat_1041M'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:59.593698Z","iopub.execute_input":"2024-04-14T00:48:59.594003Z","iopub.status.idle":"2024-04-14T00:48:59.611280Z","shell.execute_reply.started":"2024-04-14T00:48:59.593977Z","shell.execute_reply":"2024-04-14T00:48:59.610363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['lastapprcommoditycat_1041M'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:59.612426Z","iopub.execute_input":"2024-04-14T00:48:59.612753Z","iopub.status.idle":"2024-04-14T00:48:59.621895Z","shell.execute_reply.started":"2024-04-14T00:48:59.612727Z","shell.execute_reply":"2024-04-14T00:48:59.620975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['lastapprcommoditycat_1041M'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:59.623071Z","iopub.execute_input":"2024-04-14T00:48:59.623406Z","iopub.status.idle":"2024-04-14T00:48:59.641327Z","shell.execute_reply.started":"2024-04-14T00:48:59.623374Z","shell.execute_reply":"2024-04-14T00:48:59.640423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 대출 승인 여부(target)가 1인 데이터\ntarget_1_df = train_feats[train_feats['target'] == 1]\ntarget_1_grouped = target_1_df.groupby('lastapprcommoditycat_1041M').size()\n\n# 대출 승인 여부(target)가 0인 데이터\ntarget_0_df = train_feats[train_feats['target'] == 0]\ntarget_0_grouped = target_0_df.groupby('lastapprcommoditycat_1041M').size()\n\n# 그래프 그리기\nplt.figure(figsize=(12, 6))\n\n# 대출 승인 여부(target)가 1인 데이터 그래프\nplt.subplot(1, 2, 1)\nsns.barplot(x=target_1_grouped.index, y=target_1_grouped.values)\nplt.title('Loan Rejection (Target 1)')\nplt.xlabel('Commodity category')\nplt.ylabel('Frequency')\n\n# 대출 승인 여부(target)가 0인 데이터 그래프\nplt.subplot(1, 2, 2)\nsns.barplot(x=target_0_grouped.index, y=target_0_grouped.values)\nplt.title('Loan Approval (Target 0)')\nplt.xlabel('Commodity category')\nplt.ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:48:59.642319Z","iopub.execute_input":"2024-04-14T00:48:59.642561Z","iopub.status.idle":"2024-04-14T00:49:04.100469Z","shell.execute_reply.started":"2024-04-14T00:48:59.642540Z","shell.execute_reply":"2024-04-14T00:49:04.099414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df.groupby('lastapprcommoditycat_1041M').size()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:04.101855Z","iopub.execute_input":"2024-04-14T00:49:04.102149Z","iopub.status.idle":"2024-04-14T00:49:04.112617Z","shell.execute_reply.started":"2024-04-14T00:49:04.102124Z","shell.execute_reply":"2024-04-14T00:49:04.111631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 대출 승인 여부(target)가 1인 데이터\ntarget_1_df = train_feats[train_feats['target'] == 1]\ntarget_1_grouped = target_1_df.groupby('lastapprcommoditycat_1041M').size()\n\n# 대출 승인 여부(target)가 0인 데이터\ntarget_0_df = train_feats[train_feats['target'] == 0]\ntarget_0_grouped = target_0_df.groupby('lastapprcommoditycat_1041M').size()\n\n# 그래프 그리기\nplt.figure(figsize=(12, 6))\n\n# 대출 승인 여부(target)가 1인 데이터 파이 차트\nplt.subplot(1, 2, 1)\nplt.pie(target_1_grouped.values, labels=target_1_grouped.index, autopct='%1.1f%%')\nplt.title('Loan Rejection (Target 1)')\n\n# 대출 승인 여부(target)가 0인 데이터 파이 차트\nplt.subplot(1, 2, 2)\nplt.pie(target_0_grouped.values, labels=target_0_grouped.index, autopct='%1.1f%%')\nplt.title('Loan Approval (Target 0)')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:04.113835Z","iopub.execute_input":"2024-04-14T00:49:04.114117Z","iopub.status.idle":"2024-04-14T00:49:08.514298Z","shell.execute_reply.started":"2024-04-14T00:49:04.114085Z","shell.execute_reply":"2024-04-14T00:49:08.513333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 빈도 계산\ntotal = target_1_grouped.sum()\n\n# 각 소득 유형별 빈도를 전체값으로 나누고 100을 곱하여 퍼센티지로 변환\ntarget_1_percentage = (target_1_grouped / total) * 100\n\n# 결과 출력\nprint(target_1_percentage.sort_values(ascending=False).head(10))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:08.515543Z","iopub.execute_input":"2024-04-14T00:49:08.515862Z","iopub.status.idle":"2024-04-14T00:49:08.523546Z","shell.execute_reply.started":"2024-04-14T00:49:08.515836Z","shell.execute_reply":"2024-04-14T00:49:08.522549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 빈도 계산\ntotal = target_0_grouped.sum()\n\n# 각 소득 유형별 빈도를 전체값으로 나누고 100을 곱하여 퍼센티지로 변환\ntarget_0_percentage = (target_0_grouped / total) * 100\n\n# 결과 출력\nprint(target_0_percentage.sort_values(ascending=False).head(10))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:08.524669Z","iopub.execute_input":"2024-04-14T00:49:08.524921Z","iopub.status.idle":"2024-04-14T00:49:08.534897Z","shell.execute_reply.started":"2024-04-14T00:49:08.524900Z","shell.execute_reply":"2024-04-14T00:49:08.533999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df[target_1_df['lastapprcommoditycat_1041M'] != 'a55475b1']","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:08.536028Z","iopub.execute_input":"2024-04-14T00:49:08.536356Z","iopub.status.idle":"2024-04-14T00:49:08.643658Z","shell.execute_reply.started":"2024-04-14T00:49:08.536324Z","shell.execute_reply":"2024-04-14T00:49:08.642683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 대출 승인 여부(target)가 1이고 'lastapprcommoditycat_1041M' 열의 값이 'a55475b1'이 아닌 데이터 선택\ntarget_1_df_filtered = target_1_df[target_1_df['lastapprcommoditycat_1041M'] != 'a55475b1']\ntarget_1_grouped_filtered = target_1_df_filtered.groupby('lastapprcommoditycat_1041M').size()\n\n# 대출 승인 여부(target)가 0이고 'lastapprcommoditycat_1041M' 열의 값이 'a55475b1'이 아닌 데이터 선택\ntarget_0_df_filtered = target_0_df[target_0_df['lastapprcommoditycat_1041M'] != 'a55475b1']\ntarget_0_grouped_filtered = target_0_df_filtered.groupby('lastapprcommoditycat_1041M').size()\n\n# 그래프 그리기\nplt.figure(figsize=(12, 6))\n\n# 대출 승인 여부(target)가 1인 데이터 파이 차트\nplt.subplot(1, 2, 1)\nplt.pie(target_1_grouped_filtered.values, labels=target_1_grouped_filtered.index, autopct='%1.1f%%')\nplt.title('Loan Rejection (Target 1)')\n\n# 대출 승인 여부(target)가 0인 데이터 파이 차트\nplt.subplot(1, 2, 2)\nplt.pie(target_0_grouped_filtered.values, labels=target_0_grouped_filtered.index, autopct='%1.1f%%')\nplt.title('Loan Approval (Target 0)')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:08.644766Z","iopub.execute_input":"2024-04-14T00:49:08.645049Z","iopub.status.idle":"2024-04-14T00:49:11.156144Z","shell.execute_reply.started":"2024-04-14T00:49:08.645026Z","shell.execute_reply":"2024-04-14T00:49:11.155208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 빈도 계산\ntotal = target_1_grouped_filtered.sum()\n\n# 각 소득 유형별 빈도를 전체값으로 나누고 100을 곱하여 퍼센티지로 변환\ntarget_1_percentage = (target_1_grouped_filtered / total) * 100\n\n# 결과 출력\nprint(target_1_percentage.sort_values(ascending=False).head(20))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:11.157480Z","iopub.execute_input":"2024-04-14T00:49:11.158378Z","iopub.status.idle":"2024-04-14T00:49:11.166982Z","shell.execute_reply.started":"2024-04-14T00:49:11.158340Z","shell.execute_reply":"2024-04-14T00:49:11.165949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 빈도 계산\ntotal = target_0_grouped_filtered.sum()\n\n# 각 소득 유형별 빈도를 전체값으로 나누고 100을 곱하여 퍼센티지로 변환\ntarget_0_percentage = (target_0_grouped_filtered / total) * 100\n\n# 결과 출력\nprint(target_0_percentage.sort_values(ascending=False).head(20))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:11.168086Z","iopub.execute_input":"2024-04-14T00:49:11.168414Z","iopub.status.idle":"2024-04-14T00:49:11.181378Z","shell.execute_reply.started":"2024-04-14T00:49:11.168388Z","shell.execute_reply":"2024-04-14T00:49:11.180470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 대출 거절(Target 1)에 대한 바 그래프\nplt.figure(figsize=(14, 7))\nplt.subplot(1, 2, 1)\nsns.barplot(x=target_1_grouped_filtered.values, y=target_1_grouped_filtered.index)\nplt.title('Distribution of Last Approved Commodity Categories for Loan Rejection (Target 1)')\n\n# 대출 승인(Target 0)에 대한 바 그래프\nplt.subplot(1, 2, 2)\nsns.barplot(x=target_0_grouped_filtered.values, y=target_0_grouped_filtered.index)\nplt.title('Distribution of Last Approved Commodity Categories for Loan Approval (Target 0)')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:11.182795Z","iopub.execute_input":"2024-04-14T00:49:11.183435Z","iopub.status.idle":"2024-04-14T00:49:12.722324Z","shell.execute_reply.started":"2024-04-14T00:49:11.183399Z","shell.execute_reply":"2024-04-14T00:49:12.721406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(14, 6))\n\n# Target 1 (Loan Rejection) Bar Graph\nsns.barplot(y=target_1_percentage.index, x=target_1_percentage.values, ax=ax[0], palette=\"Reds_r\")\nax[0].set_title('Target 1 (Loan Rejection) Last Approved Commodity Categories Percentages')\nax[0].set_xlabel('Percentage')\nax[0].set_ylabel('Income Type')\n\n# Target 0 (Loan Approval) Bar Graph\nsns.barplot(y=target_0_percentage.index, x=target_0_percentage.values, ax=ax[1], palette=\"Greens_r\")\nax[1].set_title('Target 0 (Loan Approval) Last Approved Commodity Categories Percentages')\nax[1].set_xlabel('Percentage')\nax[1].set_ylabel('')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:12.723701Z","iopub.execute_input":"2024-04-14T00:49:12.724071Z","iopub.status.idle":"2024-04-14T00:49:14.103698Z","shell.execute_reply.started":"2024-04-14T00:49:12.724039Z","shell.execute_reply":"2024-04-14T00:49:14.102771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# interestrate_311L","metadata":{}},{"cell_type":"code","source":"train_feats['interestrate_311L'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.105116Z","iopub.execute_input":"2024-04-14T00:49:14.105489Z","iopub.status.idle":"2024-04-14T00:49:14.147388Z","shell.execute_reply.started":"2024-04-14T00:49:14.105456Z","shell.execute_reply":"2024-04-14T00:49:14.146592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 널값의 개수 계산\nnull_count = train_feats['interestrate_311L'].isnull().sum()\n\n# 전체 데이터 개수 계산\ntotal_count = len(train_feats)\n\n# 널값의 비율 계산\nnull_percentage = (null_count / total_count) * 100\n\n# 결과 출력\nprint(\"널값 비율: {:.2f}%\".format(null_percentage))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.148431Z","iopub.execute_input":"2024-04-14T00:49:14.148760Z","iopub.status.idle":"2024-04-14T00:49:14.162366Z","shell.execute_reply.started":"2024-04-14T00:49:14.148734Z","shell.execute_reply":"2024-04-14T00:49:14.161449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['interestrate_311L'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.163459Z","iopub.execute_input":"2024-04-14T00:49:14.163759Z","iopub.status.idle":"2024-04-14T00:49:14.352432Z","shell.execute_reply.started":"2024-04-14T00:49:14.163735Z","shell.execute_reply":"2024-04-14T00:49:14.351431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['interestrate_311L'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.353841Z","iopub.execute_input":"2024-04-14T00:49:14.354221Z","iopub.status.idle":"2024-04-14T00:49:14.364034Z","shell.execute_reply.started":"2024-04-14T00:49:14.354188Z","shell.execute_reply":"2024-04-14T00:49:14.363099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['interestrate_311L'].value_counts(dropna=False).sort_index()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.365411Z","iopub.execute_input":"2024-04-14T00:49:14.365856Z","iopub.status.idle":"2024-04-14T00:49:14.409578Z","shell.execute_reply.started":"2024-04-14T00:49:14.365825Z","shell.execute_reply":"2024-04-14T00:49:14.408688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['interestrate_311L'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.410719Z","iopub.execute_input":"2024-04-14T00:49:14.411002Z","iopub.status.idle":"2024-04-14T00:49:14.593511Z","shell.execute_reply.started":"2024-04-14T00:49:14.410978Z","shell.execute_reply":"2024-04-14T00:49:14.592515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target_1_df에서의 널값 비율 계산\nnull_count_target_1 = target_1_df['interestrate_311L'].isnull().sum()\ntotal_count_target_1 = len(target_1_df)\nnull_percentage_target_1 = (null_count_target_1 / total_count_target_1) * 100\n\nprint(\"target_1_df의 널값 비율: {:.2f}%\".format(null_percentage_target_1))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.594818Z","iopub.execute_input":"2024-04-14T00:49:14.595118Z","iopub.status.idle":"2024-04-14T00:49:14.601323Z","shell.execute_reply.started":"2024-04-14T00:49:14.595093Z","shell.execute_reply":"2024-04-14T00:49:14.600334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target_0_df에서의 널값 비율 계산\nnull_count_target_0 = target_0_df['interestrate_311L'].isnull().sum()\ntotal_count_target_0 = len(target_0_df)\nnull_percentage_target_0 = (null_count_target_0 / total_count_target_0) * 100\n\nprint(\"target_0_df의 널값 비율: {:.2f}%\".format(null_percentage_target_0))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.602677Z","iopub.execute_input":"2024-04-14T00:49:14.602973Z","iopub.status.idle":"2024-04-14T00:49:14.622230Z","shell.execute_reply.started":"2024-04-14T00:49:14.602945Z","shell.execute_reply":"2024-04-14T00:49:14.621337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_0_df = train_feats[train_feats['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:14.623227Z","iopub.execute_input":"2024-04-14T00:49:14.623486Z","iopub.status.idle":"2024-04-14T00:49:18.360989Z","shell.execute_reply.started":"2024-04-14T00:49:14.623464Z","shell.execute_reply":"2024-04-14T00:49:18.359924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['interestrate_311L']","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:18.362243Z","iopub.execute_input":"2024-04-14T00:49:18.362537Z","iopub.status.idle":"2024-04-14T00:49:18.370834Z","shell.execute_reply.started":"2024-04-14T00:49:18.362512Z","shell.execute_reply":"2024-04-14T00:49:18.369797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['interestrate_311L']","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:18.372078Z","iopub.execute_input":"2024-04-14T00:49:18.372415Z","iopub.status.idle":"2024-04-14T00:49:18.385825Z","shell.execute_reply.started":"2024-04-14T00:49:18.372387Z","shell.execute_reply":"2024-04-14T00:49:18.384944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the distributions of interest rates for target 0\nplt.figure(figsize=(10, 4))\nplt.hist(target_0_df['interestrate_311L'].dropna(), bins=30, alpha=0.7, label='Target 0: Loan Approval')\nplt.xlabel('Interest Rate')\nplt.ylabel('Frequency')\nplt.title('Distribution of Interest Rates for Loan Approval')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:18.387114Z","iopub.execute_input":"2024-04-14T00:49:18.387452Z","iopub.status.idle":"2024-04-14T00:49:18.852422Z","shell.execute_reply.started":"2024-04-14T00:49:18.387422Z","shell.execute_reply":"2024-04-14T00:49:18.851445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the distributions of interest rates for target 1\nplt.figure(figsize=(10, 4))\nplt.hist(target_1_df['interestrate_311L'].dropna(), bins=30, alpha=0.7, label='Target 1: Loan Rejection')\nplt.xlabel('Interest Rate')\nplt.ylabel('Frequency')\nplt.title('Distribution of Interest Rates for Loan Rejection')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:18.853801Z","iopub.execute_input":"2024-04-14T00:49:18.854501Z","iopub.status.idle":"2024-04-14T00:49:19.120346Z","shell.execute_reply.started":"2024-04-14T00:49:18.854466Z","shell.execute_reply":"2024-04-14T00:49:19.119409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 조건에 해당하는 데이터 추출\ntarget_1_interest_gt_03 = target_1_df[target_1_df['interestrate_311L'] > 0.3]\n\n# 비율 계산\npercentage_interest_gt_03 = (len(target_1_interest_gt_03) / len(target_1_df)) * 100\n\n# 결과 출력\nprint(\"Target 1에서 interest rate가 0.3보다 큰 비율: {:.2f}%\".format(percentage_interest_gt_03))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:19.121684Z","iopub.execute_input":"2024-04-14T00:49:19.122041Z","iopub.status.idle":"2024-04-14T00:49:19.206018Z","shell.execute_reply.started":"2024-04-14T00:49:19.122007Z","shell.execute_reply":"2024-04-14T00:49:19.205059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 조건에 해당하는 데이터 추출\ntarget_0_interest_gt_03 = target_0_df[target_0_df['interestrate_311L'] > 0.3]\n\n# 비율 계산\npercentage_interest_gt_03 = (len(target_0_interest_gt_03) / len(target_0_df)) * 100\n\n# 결과 출력\nprint(\"Target 0에서 interest rate가 0.3보다 큰 비율: {:.2f}%\".format(percentage_interest_gt_03))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:19.207214Z","iopub.execute_input":"2024-04-14T00:49:19.207498Z","iopub.status.idle":"2024-04-14T00:49:20.772053Z","shell.execute_reply.started":"2024-04-14T00:49:19.207474Z","shell.execute_reply":"2024-04-14T00:49:20.771046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"interestrate_311L\t\nThe interest rate of the active credit contract.\t\n활성 신용 계약의 이자율입니다.\n\ntarget 1,0별 분포 확인\ntarget 1,0에서 나타나는 특징, 차이점 확인","metadata":{}},{"cell_type":"markdown","source":"# max_person1_incometype_1044T","metadata":{}},{"cell_type":"code","source":"train_feats['max_person1_incometype_1044T'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:20.773237Z","iopub.execute_input":"2024-04-14T00:49:20.773514Z","iopub.status.idle":"2024-04-14T00:49:20.789627Z","shell.execute_reply.started":"2024-04-14T00:49:20.773490Z","shell.execute_reply":"2024-04-14T00:49:20.788691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_data = train_feats.groupby(['max_person1_incometype_1044T', 'target']).size().unstack()\ngrouped_data","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:20.790731Z","iopub.execute_input":"2024-04-14T00:49:20.791008Z","iopub.status.idle":"2024-04-14T00:49:20.853543Z","shell.execute_reply.started":"2024-04-14T00:49:20.790984Z","shell.execute_reply":"2024-04-14T00:49:20.852567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# 대출 승인 여부(target)가 1인 데이터\ntarget_1_df = train_feats[train_feats['target'] == 1]\ntarget_1_grouped = target_1_df.groupby('max_person1_incometype_1044T').size()\n\n# 대출 승인 여부(target)가 0인 데이터\ntarget_0_df = train_feats[train_feats['target'] == 0]\ntarget_0_grouped = target_0_df.groupby('max_person1_incometype_1044T').size()\n\n# 그래프 그리기\nplt.figure(figsize=(12, 6))\n\n# 대출 승인 여부(target)가 1인 데이터 그래프\nplt.subplot(1, 2, 1)\nsns.barplot(x=target_1_grouped.index, y=target_1_grouped.values)\nplt.title('Loan Rejection (Target 1)')\nplt.xlabel('Income Type')\nplt.ylabel('Frequency')\n\n# 대출 승인 여부(target)가 0인 데이터 그래프\nplt.subplot(1, 2, 2)\nsns.barplot(x=target_0_grouped.index, y=target_0_grouped.values)\nplt.title('Loan Approval (Target 0)')\nplt.xlabel('Income Type')\nplt.ylabel('Frequency')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:20.854876Z","iopub.execute_input":"2024-04-14T00:49:20.855272Z","iopub.status.idle":"2024-04-14T00:49:25.238947Z","shell.execute_reply.started":"2024-04-14T00:49:20.855238Z","shell.execute_reply":"2024-04-14T00:49:25.237932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_1_grouped = target_1_df.groupby('max_person1_incometype_1044T').size()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:25.240073Z","iopub.execute_input":"2024-04-14T00:49:25.240356Z","iopub.status.idle":"2024-04-14T00:49:25.541359Z","shell.execute_reply.started":"2024-04-14T00:49:25.240332Z","shell.execute_reply":"2024-04-14T00:49:25.540540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_grouped","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:25.542530Z","iopub.execute_input":"2024-04-14T00:49:25.542845Z","iopub.status.idle":"2024-04-14T00:49:25.550576Z","shell.execute_reply.started":"2024-04-14T00:49:25.542818Z","shell.execute_reply":"2024-04-14T00:49:25.549526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 빈도 계산\ntotal = target_1_grouped.sum()\n\n# 각 소득 유형별 빈도를 전체값으로 나누고 100을 곱하여 퍼센티지로 변환\ntarget_1_percentage = (target_1_grouped / total) * 100\n\n# 결과 출력\nprint(target_1_percentage.sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:25.551844Z","iopub.execute_input":"2024-04-14T00:49:25.552130Z","iopub.status.idle":"2024-04-14T00:49:25.562170Z","shell.execute_reply.started":"2024-04-14T00:49:25.552102Z","shell.execute_reply":"2024-04-14T00:49:25.561273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 빈도 계산\ntotal = target_0_grouped.sum()\n\n# 각 소득 유형별 빈도를 전체값으로 나누고 100을 곱하여 퍼센티지로 변환\ntarget_0_percentage = (target_0_grouped / total) * 100\n\n# 결과 출력\nprint(target_0_percentage.sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:25.563781Z","iopub.execute_input":"2024-04-14T00:49:25.564046Z","iopub.status.idle":"2024-04-14T00:49:25.574499Z","shell.execute_reply.started":"2024-04-14T00:49:25.564013Z","shell.execute_reply":"2024-04-14T00:49:25.573574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_grouped","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:25.575488Z","iopub.execute_input":"2024-04-14T00:49:25.575798Z","iopub.status.idle":"2024-04-14T00:49:25.590309Z","shell.execute_reply.started":"2024-04-14T00:49:25.575773Z","shell.execute_reply":"2024-04-14T00:49:25.589461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the results\nfig, ax = plt.subplots(1, 2, figsize=(14, 6))\n\n# Target 1 (Loan Rejection) Bar Graph\nsns.barplot(y=target_1_percentage.index, x=target_1_percentage.values, ax=ax[0], palette=\"Reds_r\")\nax[0].set_title('Target 1 (Loan Rejection) Income Type Percentages')\nax[0].set_xlabel('Percentage')\nax[0].set_ylabel('Income Type')\n\n# Target 0 (Loan Approval) Bar Graph\nsns.barplot(y=target_0_percentage.index, x=target_0_percentage.values, ax=ax[1], palette=\"Greens_r\")\nax[1].set_title('Target 0 (Loan Approval) Income Type Percentages')\nax[1].set_xlabel('Percentage')\nax[1].set_ylabel('')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:25.591478Z","iopub.execute_input":"2024-04-14T00:49:25.591854Z","iopub.status.idle":"2024-04-14T00:49:26.366114Z","shell.execute_reply.started":"2024-04-14T00:49:25.591824Z","shell.execute_reply":"2024-04-14T00:49:26.365169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"사기업 직원(PRIVATE_SECTOR_EMPLOYEE)  \n정부 급여를 받는 사람(SALARIED_GOVT)  \n은퇴 연급 수령자(RETIRED_PENSIONER)  \n고용된 사람(EMPLOYED)  \n자영업자(SELFEMPLOYED)  \n기타(OTHER)  \n장애인?(HANDICAPPED)  ","metadata":{"execution":{"iopub.status.busy":"2024-04-09T07:05:19.328586Z","iopub.execute_input":"2024-04-09T07:05:19.329485Z","iopub.status.idle":"2024-04-09T07:05:19.337891Z","shell.execute_reply.started":"2024-04-09T07:05:19.329452Z","shell.execute_reply":"2024-04-09T07:05:19.336535Z"}}},{"cell_type":"markdown","source":"소득 유형이 대출 거절 및 승인에 영향을 미치는 방식에 있어서 차이가 존재한다.\n\ntarget 1 :   \n사기업 직원 > 고용된 사람 > 정부 급여 수령자 > 은퇴 연금 수령자\n\ntarget 0 :   \n사기업 직원이 높은 비율이긴 하지만 target 1 에 비해 비율이 낮음  \n정부 급여 수령자와 은퇴 연금 수령자 비율이 높게 나타남(소득이 안정적인 사람)\n           \n=> 소득이 안정적인지에 따라 target에 영향을 주는 것으로 확인된다.","metadata":{}},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_0_df = train_feats[train_feats['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:26.367382Z","iopub.execute_input":"2024-04-14T00:49:26.367703Z","iopub.status.idle":"2024-04-14T00:49:29.438276Z","shell.execute_reply.started":"2024-04-14T00:49:26.367676Z","shell.execute_reply":"2024-04-14T00:49:29.437259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lastapprcommoditycat_1041M 별 interestrate_311L","metadata":{}},{"cell_type":"code","source":"train_feats['interestrate_311L'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:29.439561Z","iopub.execute_input":"2024-04-14T00:49:29.439881Z","iopub.status.idle":"2024-04-14T00:49:29.479358Z","shell.execute_reply.started":"2024-04-14T00:49:29.439855Z","shell.execute_reply":"2024-04-14T00:49:29.478488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['lastapprcommoditycat_1041M'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:29.491048Z","iopub.execute_input":"2024-04-14T00:49:29.491319Z","iopub.status.idle":"2024-04-14T00:49:29.507003Z","shell.execute_reply.started":"2024-04-14T00:49:29.491297Z","shell.execute_reply":"2024-04-14T00:49:29.506083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_0_df = train_feats[train_feats['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:29.507973Z","iopub.execute_input":"2024-04-14T00:49:29.508235Z","iopub.status.idle":"2024-04-14T00:49:32.575702Z","shell.execute_reply.started":"2024-04-14T00:49:29.508212Z","shell.execute_reply":"2024-04-14T00:49:32.574876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'lastapprcommoditycat_1041M' 카테고리별로 이자율 'interestrate_311L'의 평균 계산\ninterest_rate_by_category = target_1_df.groupby('lastapprcommoditycat_1041M')['interestrate_311L'].describe()\ninterest_rate_by_category_sorted_1 = interest_rate_by_category.sort_values(by='count', ascending=False)\ninterest_rate_by_category_sorted_1.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:32.577053Z","iopub.execute_input":"2024-04-14T00:49:32.577897Z","iopub.status.idle":"2024-04-14T00:49:32.665133Z","shell.execute_reply.started":"2024-04-14T00:49:32.577861Z","shell.execute_reply":"2024-04-14T00:49:32.664240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# interest_rate_by_category_sorted_1.to_csv('interest_rate_by_category_sorted_1.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:32.666481Z","iopub.execute_input":"2024-04-14T00:49:32.666876Z","iopub.status.idle":"2024-04-14T00:49:32.671328Z","shell.execute_reply.started":"2024-04-14T00:49:32.666843Z","shell.execute_reply":"2024-04-14T00:49:32.670276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'lastapprcommoditycat_1041M' 카테고리별로 이자율 'interestrate_311L'의 평균 계산\ninterest_rate_by_category = target_0_df.groupby('lastapprcommoditycat_1041M')['interestrate_311L'].describe()\ninterest_rate_by_category_sorted_0 = interest_rate_by_category.sort_values(by='count', ascending=False)\ninterest_rate_by_category_sorted_0.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:32.672536Z","iopub.execute_input":"2024-04-14T00:49:32.672827Z","iopub.status.idle":"2024-04-14T00:49:32.972828Z","shell.execute_reply.started":"2024-04-14T00:49:32.672804Z","shell.execute_reply":"2024-04-14T00:49:32.971691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# interest_rate_by_category_sorted_0.to_csv('interest_rate_by_category_sorted_0.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:32.974083Z","iopub.execute_input":"2024-04-14T00:49:32.974362Z","iopub.status.idle":"2024-04-14T00:49:32.978545Z","shell.execute_reply.started":"2024-04-14T00:49:32.974339Z","shell.execute_reply":"2024-04-14T00:49:32.977643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target 1에서 상위 10개 대출 유형에 대한 데이터 필터링\ntop10_categories_1 = interest_rate_by_category_sorted_1.head(10).index\nfiltered_data_1 = target_1_df[target_1_df['lastapprcommoditycat_1041M'].isin(top10_categories_1)]\n\n# 박스플랏 그리기\nplt.figure(figsize=(12, 8))\nsns.boxplot(x='interestrate_311L', y='lastapprcommoditycat_1041M', data=filtered_data_1, order=top10_categories_1)\nplt.title('Interest Rate Distribution by Commodity Category for Target 1 (Top 10)')\nplt.xlabel('Interest Rate')\nplt.ylabel('Commodity Category')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:32.979756Z","iopub.execute_input":"2024-04-14T00:49:32.980080Z","iopub.status.idle":"2024-04-14T00:49:33.490672Z","shell.execute_reply.started":"2024-04-14T00:49:32.980043Z","shell.execute_reply":"2024-04-14T00:49:33.489700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# target 0에서 상위 10개 대출 유형에 대한 데이터 필터링\ntop10_categories_0 = interest_rate_by_category_sorted_0.head(10).index\nfiltered_data_0 = target_0_df[target_0_df['lastapprcommoditycat_1041M'].isin(top10_categories_0)]\n\n# 박스플랏 그리기\nplt.figure(figsize=(12, 8))\nsns.boxplot(x='interestrate_311L', y='lastapprcommoditycat_1041M', data=filtered_data_0, order=top10_categories_0)\nplt.title('Interest Rate Distribution by Commodity Category for Target 0 (Top 10)')\nplt.xlabel('Interest Rate')\nplt.ylabel('Commodity Category')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:33.492074Z","iopub.execute_input":"2024-04-14T00:49:33.492754Z","iopub.status.idle":"2024-04-14T00:49:36.819332Z","shell.execute_reply.started":"2024-04-14T00:49:33.492711Z","shell.execute_reply":"2024-04-14T00:49:36.818287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'lastapprcommoditycat_1041M' 카테고리별로 이자율 'interestrate_311L'의 평균 계산\ninterest_rate_by_category = train_feats.groupby('lastapprcommoditycat_1041M')['interestrate_311L'].describe()\ninterest_rate_by_category","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:36.820811Z","iopub.execute_input":"2024-04-14T00:49:36.821229Z","iopub.status.idle":"2024-04-14T00:49:37.177249Z","shell.execute_reply.started":"2024-04-14T00:49:36.821190Z","shell.execute_reply":"2024-04-14T00:49:37.176223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lastrejectcommoditycat_161M","metadata":{}},{"cell_type":"code","source":"train_feats['lastrejectcommoditycat_161M'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:37.178506Z","iopub.execute_input":"2024-04-14T00:49:37.178924Z","iopub.status.idle":"2024-04-14T00:49:37.199734Z","shell.execute_reply.started":"2024-04-14T00:49:37.178895Z","shell.execute_reply":"2024-04-14T00:49:37.198705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interest_rate_by_category = train_feats.groupby('lastrejectcommoditycat_161M')['interestrate_311L'].describe()\ninterest_rate_by_category","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:37.201093Z","iopub.execute_input":"2024-04-14T00:49:37.201832Z","iopub.status.idle":"2024-04-14T00:49:37.571054Z","shell.execute_reply.started":"2024-04-14T00:49:37.201790Z","shell.execute_reply":"2024-04-14T00:49:37.569881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interest_rate_by_category = target_1_df.groupby('lastrejectcommoditycat_161M')['interestrate_311L'].describe()\ninterest_rate_by_category_sorted_1 = interest_rate_by_category.sort_values(by='count', ascending=False)\ninterest_rate_by_category_sorted_1.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:37.573034Z","iopub.execute_input":"2024-04-14T00:49:37.573569Z","iopub.status.idle":"2024-04-14T00:49:37.661303Z","shell.execute_reply.started":"2024-04-14T00:49:37.573529Z","shell.execute_reply":"2024-04-14T00:49:37.660329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target 1에서 상위 10개 대출 유형에 대한 데이터 필터링\ntop10_categories_1 = interest_rate_by_category_sorted_1.head(10).index\nfiltered_data_1 = target_1_df[target_1_df['lastrejectcommoditycat_161M'].isin(top10_categories_1)]\n\n# 박스플랏 그리기\nplt.figure(figsize=(12, 8))\nsns.boxplot(x='interestrate_311L', y='lastrejectcommoditycat_161M', data=filtered_data_1, order=top10_categories_1)\nplt.title('Interest Rate Distribution by Last Reject Commodity Category for Target 1 (Top 10)')\nplt.xlabel('Interest Rate')\nplt.ylabel('Last Reject Commodity Category')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:37.662374Z","iopub.execute_input":"2024-04-14T00:49:37.662683Z","iopub.status.idle":"2024-04-14T00:49:38.218485Z","shell.execute_reply.started":"2024-04-14T00:49:37.662657Z","shell.execute_reply":"2024-04-14T00:49:38.217549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interest_rate_by_category = target_0_df.groupby('lastrejectcommoditycat_161M')['interestrate_311L'].describe()\ninterest_rate_by_category_sorted_0 = interest_rate_by_category.sort_values(by='count', ascending=False)\ninterest_rate_by_category_sorted_0.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:38.219956Z","iopub.execute_input":"2024-04-14T00:49:38.220782Z","iopub.status.idle":"2024-04-14T00:49:38.527860Z","shell.execute_reply.started":"2024-04-14T00:49:38.220745Z","shell.execute_reply":"2024-04-14T00:49:38.526874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target 1에서 상위 10개 대출 유형에 대한 데이터 필터링\ntop10_categories_0 = interest_rate_by_category_sorted_0.head(10).index\nfiltered_data_0 = target_0_df[target_0_df['lastrejectcommoditycat_161M'].isin(top10_categories_0)]\n\n# 박스플랏 그리기\nplt.figure(figsize=(12, 8))\nsns.boxplot(x='interestrate_311L', y='lastrejectcommoditycat_161M', data=filtered_data_0, order=top10_categories_0)\nplt.title('Interest Rate Distribution by Last Reject Commodity Category for Target 0 (Top 10)')\nplt.xlabel('Interest Rate')\nplt.ylabel('Last Reject Commodity Category')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:38.529128Z","iopub.execute_input":"2024-04-14T00:49:38.529441Z","iopub.status.idle":"2024-04-14T00:49:43.034458Z","shell.execute_reply.started":"2024-04-14T00:49:38.529415Z","shell.execute_reply":"2024-04-14T00:49:43.033394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# last_person1_relationshiptoclient_642T","metadata":{}},{"cell_type":"code","source":"train_feats['last_person1_relationshiptoclient_642T']","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.035527Z","iopub.execute_input":"2024-04-14T00:49:43.035826Z","iopub.status.idle":"2024-04-14T00:49:43.045791Z","shell.execute_reply.started":"2024-04-14T00:49:43.035800Z","shell.execute_reply":"2024-04-14T00:49:43.044907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats['last_person1_relationshiptoclient_642T'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.047061Z","iopub.execute_input":"2024-04-14T00:49:43.047418Z","iopub.status.idle":"2024-04-14T00:49:43.072129Z","shell.execute_reply.started":"2024-04-14T00:49:43.047393Z","shell.execute_reply":"2024-04-14T00:49:43.071195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['last_person1_relationshiptoclient_642T'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.073341Z","iopub.execute_input":"2024-04-14T00:49:43.074060Z","iopub.status.idle":"2024-04-14T00:49:43.094134Z","shell.execute_reply.started":"2024-04-14T00:49:43.074020Z","shell.execute_reply":"2024-04-14T00:49:43.093188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 데이터 중 널 값의 비율 계산\nnull_ratio = target_0_df['last_person1_relationshiptoclient_642T'].isnull().mean() * 100\n\n# 널 값 비율 출력\nprint(f\"Null value ratio in 'last_person1_relationshiptoclient_642T': {null_ratio:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.095232Z","iopub.execute_input":"2024-04-14T00:49:43.095536Z","iopub.status.idle":"2024-04-14T00:49:43.102273Z","shell.execute_reply.started":"2024-04-14T00:49:43.095511Z","shell.execute_reply":"2024-04-14T00:49:43.101355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['last_person1_relationshiptoclient_642T'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.103458Z","iopub.execute_input":"2024-04-14T00:49:43.103801Z","iopub.status.idle":"2024-04-14T00:49:43.115320Z","shell.execute_reply.started":"2024-04-14T00:49:43.103773Z","shell.execute_reply":"2024-04-14T00:49:43.114327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전체 데이터 중 널 값의 비율 계산\nnull_ratio = target_1_df['last_person1_relationshiptoclient_642T'].isnull().mean() * 100\n\n# 널 값 비율 출력\nprint(f\"Null value ratio in 'last_person1_relationshiptoclient_642T': {null_ratio:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.116384Z","iopub.execute_input":"2024-04-14T00:49:43.116700Z","iopub.status.idle":"2024-04-14T00:49:43.122731Z","shell.execute_reply.started":"2024-04-14T00:49:43.116665Z","shell.execute_reply":"2024-04-14T00:49:43.121689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 널 값이 있는 행 제외하고 데이터프레임 생성\ntrain_feats = train_feats.dropna(subset=['last_person1_relationshiptoclient_642T'])\ntrain_feats","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:43.124051Z","iopub.execute_input":"2024-04-14T00:49:43.124368Z","iopub.status.idle":"2024-04-14T00:49:44.507988Z","shell.execute_reply.started":"2024-04-14T00:49:43.124333Z","shell.execute_reply":"2024-04-14T00:49:44.506896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df = train_feats[train_feats['target'] == 1]\ntarget_0_df = train_feats[train_feats['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:44.509367Z","iopub.execute_input":"2024-04-14T00:49:44.509694Z","iopub.status.idle":"2024-04-14T00:49:45.735194Z","shell.execute_reply.started":"2024-04-14T00:49:44.509658Z","shell.execute_reply":"2024-04-14T00:49:45.734112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_df['last_person1_relationshiptoclient_642T'].value_counts(normalize=True, dropna=False) * 100","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:45.736375Z","iopub.execute_input":"2024-04-14T00:49:45.736676Z","iopub.status.idle":"2024-04-14T00:49:45.746764Z","shell.execute_reply.started":"2024-04-14T00:49:45.736627Z","shell.execute_reply":"2024-04-14T00:49:45.745716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_df['last_person1_relationshiptoclient_642T'].value_counts(normalize=True, dropna=False) * 100","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:45.748171Z","iopub.execute_input":"2024-04-14T00:49:45.748451Z","iopub.status.idle":"2024-04-14T00:49:45.763705Z","shell.execute_reply.started":"2024-04-14T00:49:45.748427Z","shell.execute_reply":"2024-04-14T00:49:45.762668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# 시각화를 위한 데이터 준비\ntarget_1_value_counts = target_1_df['last_person1_relationshiptoclient_642T'].value_counts(normalize=True, dropna=False) * 100\ntarget_0_value_counts = target_0_df['last_person1_relationshiptoclient_642T'].value_counts(normalize=True, dropna=False) * 100\n\n# 시각화\nplt.figure(figsize=(10, 6))\n\n# target_1 시각화\nplt.bar(target_1_value_counts.index, target_1_value_counts.values, color='b', alpha=0.5, label='Target 1')\n\n# target_0 시각화\nplt.bar(target_0_value_counts.index, target_0_value_counts.values, color='r', alpha=0.5, label='Target 0')\n\nplt.xlabel('Value')\nplt.ylabel('Percentage')\nplt.title('Value Distribution by Target')\nplt.legend()\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T04:05:53.196365Z","iopub.execute_input":"2024-04-14T04:05:53.196731Z","iopub.status.idle":"2024-04-14T04:05:53.620954Z","shell.execute_reply.started":"2024-04-14T04:05:53.196696Z","shell.execute_reply":"2024-04-14T04:05:53.620064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def pickle_dump(obj, path):\n#     #打开指定的路径path,binary write(二进制写入)\n#     with open(path, mode=\"wb\") as f:\n#         #将obj对象保存到f,使用协议版本4进行序列化\n#         dill.dump(obj, f, protocol=4)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:46.164496Z","iopub.execute_input":"2024-04-14T00:49:46.164803Z","iopub.status.idle":"2024-04-14T00:49:46.169011Z","shell.execute_reply.started":"2024-04-14T00:49:46.164777Z","shell.execute_reply":"2024-04-14T00:49:46.168064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def fit_and_predict(model,train_feats=train_feats,test_feats=test_feats,name=0):\n#     X=train_feats[choose_cols].copy()\n#     y=train_feats[Config.TARGET_NAME].copy()\n#     test_X=test_feats[choose_cols].copy()\n#     del train_feats,test_feats\n#     gc.collect()#手动触发垃圾回收,强制回收由垃圾回收器标记为未使用的内存\n#     oof_pred_pro=np.zeros((len(X),2))\n#     test_pred_pro=np.zeros((Config.num_folds,len(test_X),2))\n\n#     #10折交叉验证\n#     skf = StratifiedKFold(n_splits=Config.num_folds,random_state=Config.seed, shuffle=True)\n\n#     for fold, (train_index, valid_index) in (enumerate(skf.split(X, y.astype(str)))):\n#         print(f\"fold:{fold}\")\n\n#         X_train, X_valid = X.iloc[train_index], X.iloc[valid_index]\n#         y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n        \n#         model.fit(X_train,y_train,eval_set=[(X_valid, y_valid)],\n#                   callbacks=[log_evaluation(100),early_stopping(100)]\n#                  )\n        \n#         oof_pred_pro[valid_index]=model.predict_proba(X_valid)\n#         #将数据分批次进行预测.\n#         for idx in range(0,len(test_X),Config.batch_size):\n#             test_pred_pro[fold][idx:idx+Config.batch_size]=model.predict_proba(test_X[idx:idx+Config.batch_size])\n#         pickle_dump(model, f'/kaggle/working/lgb_fold{Config.num_folds*name+fold}.model') #保存训练好的模型   \n#     gini=2*roc_auc_score(y.values,oof_pred_pro[:,1])-1\n#     print(f\"mean_gini:{gini}\")\n    \n#     return oof_pred_pro,test_pred_pro","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:46.170338Z","iopub.execute_input":"2024-04-14T00:49:46.170701Z","iopub.status.idle":"2024-04-14T00:49:46.180525Z","shell.execute_reply.started":"2024-04-14T00:49:46.170669Z","shell.execute_reply":"2024-04-14T00:49:46.179547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #参数来源:https://www.kaggle.com/code/daviddirethucus/home-credit-risk-lightgbm\n# lgb_params={\n#     \"boosting_type\": \"gbdt\",\n#     \"objective\": \"binary\",\n#     \"metric\": \"auc\",\n#     \"max_depth\": 10,\n#     \"learning_rate\": 0.05,\n#     \"n_estimators\": 2560,\n#     \"colsample_bytree\": 0.8,\n#     \"colsample_bynode\": 0.8,\n#     \"verbose\": -1,\n#     \"random_state\": Config.seed,\n#     \"reg_alpha\": 0.1,\n#     \"reg_lambda\": 10,\n#     \"extra_trees\":True,\n#     'num_leaves':64,\n#     \"verbose\": -1,\n#     \"max_bin\":245,\n#     'device':'gpu',\n#     }\n\n# lgb_oof_pred_pro,lgb_test_pred_pro=fit_and_predict(model= LGBMClassifier(**lgb_params),\n#                                                   train_feats=train_feats,test_feats=test_feats,name=0\n#                                                   )","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:46.181976Z","iopub.execute_input":"2024-04-14T00:49:46.182825Z","iopub.status.idle":"2024-04-14T00:49:46.195254Z","shell.execute_reply.started":"2024-04-14T00:49:46.182761Z","shell.execute_reply":"2024-04-14T00:49:46.194247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_preds=lgb_test_pred_pro.mean(axis=0)[:,1]\n# submission=pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\n# submission['score']=np.clip(np.nan_to_num(test_preds,nan=0.314),0,1)\n# submission.to_csv(\"submission.csv\",index=None)\n# submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:46.196309Z","iopub.execute_input":"2024-04-14T00:49:46.196609Z","iopub.status.idle":"2024-04-14T00:49:46.207390Z","shell.execute_reply.started":"2024-04-14T00:49:46.196576Z","shell.execute_reply":"2024-04-14T00:49:46.206453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#这里就是个示例代码,有很多特征,然后还有每折交叉验证的特征重要性,最后得出小于margin的不重要的特征.\n# import re\n# import numpy as np\n# total_features=['a','b','c']\n# text=\"\"\"[1 22 23] [3 2 30]\n# \"\"\"\n# nums=re.findall(r'\\d+',text)\n# nums=np.array([int(num) for num in nums])\n# margin=5#如果一个特征在2000个迭代器中选择的次数少于5次,就说明这个特征不重要.\n# useless_cols=[]\n# for i in range(len(nums)):\n#     if nums[i]<margin:\n#         feats=total_features[i%len(total_features)]\n#         if feats not in useless_cols:\n#             useless_cols.append(feats)\n# print(f\"len(useless_cols):{len(useless_cols)},useless_cols:{useless_cols}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T00:49:46.208496Z","iopub.execute_input":"2024-04-14T00:49:46.209327Z","iopub.status.idle":"2024-04-14T00:49:46.216434Z","shell.execute_reply.started":"2024-04-14T00:49:46.209293Z","shell.execute_reply":"2024-04-14T00:49:46.215656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Future work to be done:\n\n### Selection and construction of features\n\n### ensemble more models\n\n### ……","metadata":{}}]}