{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7941325,"sourceType":"datasetVersion","datasetId":4506020}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Home Credit Model and Submission","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings, os, gc, joblib\nfrom pprint import pprint\nimport scikitplot as skplt\nimport lightgbm as lgb\nfrom sklearn import metrics\nfrom functools import reduce\nfrom sklearn.metrics import accuracy_score, roc_auc_score, confusion_matrix, ConfusionMatrixDisplay, classification_report\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV, StratifiedGroupKFold\nfrom contextlib import suppress\n# from imblearn.under_sampling import NearMiss\n# from imblearn.over_sampling import SMOTE","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:43.908021Z","iopub.execute_input":"2024-03-28T15:37:43.908516Z","iopub.status.idle":"2024-03-28T15:37:46.199443Z","shell.execute_reply.started":"2024-03-28T15:37:43.908478Z","shell.execute_reply":"2024-03-28T15:37:46.198019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pathway = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\n\ndef set_table_dtypes(df: pl.DataFrame)-> pl.DataFrame:\n    for col in df.columns:\n        # Cast Transform DPD (Days past due, P) and Transform Amount (A) as Float64\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n        # Cast Transform date (D) as Date\n        if col[-1] in (\"D\"):\n            df = df.with_columns(pl.col(col).cast(pl.Date).alias(col))\n        # Cast aggregated columns as Float64, tried combining sum and max, but did not work correctly\n        if col[-4:-1] in ('_sum'):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n        if col[-4:-1] in ('_max'):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n    return df\n\ndef convert_strings(df: pl.DataFrame) -> pl.DataFrame:\n    for col in df.columns:\n        if df[col].dtype == pl.Utf8:\n            df = df.with_columns(pl.col(col).cast(pl.Categorical))\n    return df\n\n# Changed this function to work for Pandas\ndef missing_values(df, threshold = 0.0):\n    for col in df.columns:\n        decimal = (pd.isnull(test[col]).sum())/(len(test[col]))\n        if decimal > threshold:                                         \n            print(f\"{col}: {decimal}\")\n\n# Impute numeric columns with the median and cat with mode\ndef imputer(df:pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:\n        if df[col].dtype == 'float64':\n            df[col] = df[col].fillna(df[col].median())\n        if df[col].dtype.name in ['category','object'] and df[col].isnull().any():\n            mode_without_nan = df[col].dropna().mode().values[0]\n            df[col] = df[col].fillna(mode_without_nan)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.205262Z","iopub.execute_input":"2024-03-28T15:37:46.205681Z","iopub.status.idle":"2024-03-28T15:37:46.343213Z","shell.execute_reply.started":"2024-03-28T15:37:46.205648Z","shell.execute_reply":"2024-03-28T15:37:46.342013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate joined test data \n#### Part 1\nPreprocessed the training csvs in the same way, lists of columns that are dropped were manually taken from what was dropped in training based on excessive missing values. Used .drop(errors='ignore') to handle situations where hidden test set has different columns. ","metadata":{}},{"cell_type":"code","source":"test_basetable = pl.read_csv(pathway + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [pl.read_csv(pathway + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes)\n    ], how=\"vertical_relaxed\")\ntest_static_cb=pl.read_csv(pathway + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1=pl.read_csv(pathway +  \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes)\ntest_credit_bureau_b_2=pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.344955Z","iopub.execute_input":"2024-03-28T15:37:46.345671Z","iopub.status.idle":"2024-03-28T15:37:46.411779Z","shell.execute_reply.started":"2024-03-28T15:37:46.345610Z","shell.execute_reply":"2024-03-28T15:37:46.410678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Additional depth=1 files\ntest_other_1 = pl.read_csv(pathway + \"csv_files/test/test_other_1.csv\").pipe(set_table_dtypes)\n\ntest_credit_bureau_b_1 = pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_b_1.csv\").pipe(set_table_dtypes)\n\ntest_deposit_1 = pl.read_csv(pathway + \"csv_files/test/test_deposit_1.csv\").pipe(set_table_dtypes)\n\n# test_debitcard_1 = pl.read_csv(pathway + \"csv_files/test/test_debitcard_1.csv\").pipe(set_table_dtypes)\ntest_basetable = test_basetable.with_columns(pl.col('date_decision').cast(pl.Date))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.415731Z","iopub.execute_input":"2024-03-28T15:37:46.416303Z","iopub.status.idle":"2024-03-28T15:37:46.432625Z","shell.execute_reply.started":"2024-03-28T15:37:46.416255Z","shell.execute_reply":"2024-03-28T15:37:46.431168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#use aggregation functions in tables with depth >=1\n\ntest_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").sum().alias(\"mainoccupationinc_384A_sum\"))\n\n#num_group1=0 represents the person who applied for the loan\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"incometype_1044T\", \"birth_259D\",\n    \"empl_employedfrom_271D\",\"empl_industry_691L\",\"familystate_447L\",\"sex_738L\",\"type_25L\",\n    \"safeguarantyflag_411L\",\"empl_employedtotal_800L\",\"role_1084L\"]).filter(\n    pl.col(\"num_group1\")==0).drop(\"num_group1\")\n\n#we now have num_group1 and num_group2, so aggregate again\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").sum().alias(\"pmts_pmtsoverdue_635A_sum\"),\n    pl.col(\"pmts_dpdvalue_108P\").sum().alias(\"pmts_dpdvalue_108P_sum\"))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.434777Z","iopub.execute_input":"2024-03-28T15:37:46.435346Z","iopub.status.idle":"2024-03-28T15:37:46.445766Z","shell.execute_reply.started":"2024-03-28T15:37:46.435304Z","shell.execute_reply":"2024-03-28T15:37:46.444693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Additional aggregation for depth=1 files\ntest_other_1_feats = test_other_1.group_by(\"case_id\").agg(\n    pl.col(\"amtdebitincoming_4809443A\").sum().alias(\"amtdebitincoming_4809443A_sum\"),\n    pl.col(\"amtdebitoutgoing_4809440A\").sum().alias(\"amtdebitoutgoing_4809440A_sum\"),\n    pl.col(\"amtdepositbalance_4809441A\").sum().alias(\"amtdepositbalance_4809441A_sum\"),\n    pl.col(\"amtdepositincoming_4809444A\").sum().alias(\"amtdepositincoming_4809444A_sum\"),\n    pl.col(\"amtdepositoutgoing_4809442A\").sum().alias(\"amtdepositoutgoing_4809442A_sum\"))\n\ntest_credit_bureau_b_1_feats = test_credit_bureau_b_1.group_by(\"case_id\").agg(\n    pl.col(\"amount_1115A\").sum().alias(\"amount_1115A_sum\"),\n    pl.col(\"credquantity_1099L\").sum().alias(\"credquantity_1099L_sum\"),\n    pl.col(\"credquantity_984L\").sum().alias(\"credquantity_984L_sum\"),\n    pl.col(\"debtpastduevalue_732A\").sum().alias(\"debtpastduevalue_732A_sum\"),\n    pl.col(\"debtvalue_227A\").sum().alias(\"debtvalue_227A_sum\"),\n    pl.col(\"dpd_550P\").sum().alias(\"dpd_550P_sum\"),\n    pl.col(\"dpd_733P\").sum().alias(\"dpd_733P_sum\"),\n    pl.col(\"dpdmax_851P\").max().alias(\"dpdmax_851P_max\"),\n    pl.col(\"installmentamount_644A\").sum().alias(\"installmentamount_644A_sum\"),\n    pl.col(\"installmentamount_833A\").sum().alias(\"installmentamount_833A_sum\"),\n    pl.col(\"instlamount_892A\").sum().alias(\"instlamount_892A_sum\"),\n    pl.col(\"interestrateyearly_538L\").max().alias(\"interestrateyearly_538L_max\"),\n    pl.col(\"maxdebtpduevalodued_3940955A\").max().alias(\"maxdebtpduevalodued_3940955A_max\"),\n    pl.col(\"numberofinstls_810L\").sum().alias(\"numberofinstls_810L_sum\"),\n    pl.col(\"overdueamountmax_950A\").max().alias(\"overdueamountmax_950A_max\"),\n    pl.col(\"pmtdaysoverdue_1135P\").sum().alias(\"pmtdaysoverdue_1135P_sum\"),\n    pl.col(\"pmtnumpending_403L\").sum().alias(\"pmtnumpending_403L_sum\"),\n    pl.col(\"residualamount_3940956A\").sum().alias(\"residualamount_3940956A_sum\"),\n    pl.col(\"totalamount_503A\").sum().alias(\"totalamount_503A_sum\"),\n    pl.col(\"totalamount_881A\").sum().alias(\"totalamount_881A_sum\"))\n\ntest_deposit_1_feats = test_deposit_1.group_by(\"case_id\").agg(\n    pl.col(\"amount_416A\").sum().alias(\"amount_416A_sum\"))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.447267Z","iopub.execute_input":"2024-03-28T15:37:46.447909Z","iopub.status.idle":"2024-03-28T15:37:46.470860Z","shell.execute_reply.started":"2024-03-28T15:37:46.447873Z","shell.execute_reply":"2024-03-28T15:37:46.469737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# join all tables/columns together\n\njoin_data1 = test_basetable.join(test_static, how=\"left\", on=\"case_id\"\n).join(test_static_cb, how=\"left\", on=\"case_id\"\n).join(test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n).join(test_other_1_feats, how=\"left\", on=\"case_id\"\n).join(test_credit_bureau_b_1_feats, how=\"left\", on=\"case_id\"\n).join(test_deposit_1_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.472658Z","iopub.execute_input":"2024-03-28T15:37:46.473915Z","iopub.status.idle":"2024-03-28T15:37:46.494496Z","shell.execute_reply.started":"2024-03-28T15:37:46.473856Z","shell.execute_reply":"2024-03-28T15:37:46.492872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# After merge, convert back to pandas for errors='ignore' functionality \njoin_data1 = join_data1.to_pandas()\n\ndrop_cols = ['avgdbddpdlast3m_4187120P', 'avgdbdtollast24m_4525197P', 'avglnamtstart24m_4525187A', 'avgpmtlast12m_4525200A', 'bankacctype_710L', 'cardtype_51L', 'clientscnt_136L', 'datelastinstal40dpd_247D', 'dtlastpmtallstes_4499206D', 'equalitydataagreement_891L', 'equalityempfrom_62L', 'inittransactionamount_650A', 'interestrategrace_34L', 'isbidproductrequest_292L', 'isdebitcard_729L', 'lastdelinqdate_224D', 'lastdependentsnum_448L', 'lastotherinc_902A', 'lastotherlnsexpense_631A', 'lastrepayingdate_696D', 'maxannuity_4075009A', 'maxdbddpdlast1m_3658939P', 'maxlnamtstart6m_4525199A', 'maxpmtlast3m_4525190A', 'mindbdtollast24m_4525191P', 'payvacationpostpone_4187118D', 'totinstallast1m_4525188A', 'typesuite_864L', 'validfrom_1069D', 'assignmentdate_238D', 'assignmentdate_4527235D', 'assignmentdate_4955616D', 'birthdate_574D', 'contractssum_5085716L', 'dateofbirth_342D', 'for3years_128L', 'for3years_504L', 'for3years_584L', 'formonth_118L', 'formonth_206L', 'formonth_535L', 'forquarter_1017L', 'forquarter_462L', 'forquarter_634L', 'fortoday_1092L', 'forweek_1077L', 'forweek_528L', 'forweek_601L', 'foryear_618L', 'foryear_818L', 'foryear_850L', 'pmtaverage_3A', 'pmtaverage_4527227A', 'pmtaverage_4955615A', 'pmtcount_4527229L', 'pmtcount_4955617L', 'pmtcount_693L', 'pmtscount_423L', 'pmtssum_45A', 'responsedate_4917613D', 'riskassesment_302T', 'riskassesment_940T', 'empl_employedfrom_271D', 'empl_industry_691L', 'empl_employedtotal_800L', 'pmts_pmtsoverdue_635A_sum', 'pmts_dpdvalue_108P_sum', 'amtdebitincoming_4809443A_sum', 'amtdebitoutgoing_4809440A_sum', 'amtdepositbalance_4809441A_sum', 'amtdepositincoming_4809444A_sum', 'amtdepositoutgoing_4809442A_sum', 'amount_1115A_sum', 'credquantity_1099L_sum', 'credquantity_984L_sum', 'debtpastduevalue_732A_sum', 'debtvalue_227A_sum', 'dpd_550P_sum', 'dpd_733P_sum', 'dpdmax_851P_max', 'installmentamount_644A_sum', 'installmentamount_833A_sum', 'instlamount_892A_sum', 'interestrateyearly_538L_max', 'maxdebtpduevalodued_3940955A_max', 'numberofinstls_810L_sum', 'overdueamountmax_950A_max', 'pmtdaysoverdue_1135P_sum', 'pmtnumpending_403L_sum', 'residualamount_3940956A_sum', 'totalamount_503A_sum', 'totalamount_881A_sum', 'amount_416A_sum']\njoin_data1 = join_data1.drop(drop_cols, axis=1, errors='ignore')\njoin_data1.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.495976Z","iopub.execute_input":"2024-03-28T15:37:46.496381Z","iopub.status.idle":"2024-03-28T15:37:46.532879Z","shell.execute_reply.started":"2024-03-28T15:37:46.496349Z","shell.execute_reply":"2024-03-28T15:37:46.531532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_static, test_static_cb, test_person_1, test_credit_bureau_b_2, test_other_1,test_credit_bureau_b_1,test_deposit_1\ndel test_person_1_feats_1, test_person_1_feats_2, test_credit_bureau_b_2_feats, test_other_1_feats, test_credit_bureau_b_1_feats, test_deposit_1_feats   \ngc.collect()  ","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.534474Z","iopub.execute_input":"2024-03-28T15:37:46.535425Z","iopub.status.idle":"2024-03-28T15:37:46.664362Z","shell.execute_reply.started":"2024-03-28T15:37:46.535389Z","shell.execute_reply":"2024-03-28T15:37:46.663011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Part 2","metadata":{}},{"cell_type":"code","source":"# Additional testing data, depth = 1\ntest_applprev_1 = pl.concat(\n    [pl.read_csv(pathway + \"csv_files/test/test_applprev_1_0.csv\").pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_applprev_1_1.csv\").pipe(set_table_dtypes) \n    ], how=\"vertical_relaxed\")\n\ntest_tax_registry_a_1 = pl.read_csv(pathway + \"csv_files/test/test_tax_registry_a_1.csv\").pipe(set_table_dtypes)\ntest_tax_registry_b_1 = pl.read_csv(pathway + \"csv_files/test/test_tax_registry_b_1.csv\").pipe(set_table_dtypes)    \ntest_tax_registry_c_1 = pl.read_csv(pathway + \"csv_files/test/test_tax_registry_c_1.csv\").pipe(set_table_dtypes)\n    \ntest_credit_bureau_a_1 = pl.concat(\n    [pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_1_0.csv\").pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_1_1.csv\").pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_1_2.csv\").pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_1_3.csv\").pipe(set_table_dtypes),\n    ], how=\"vertical_relaxed\")","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.666406Z","iopub.execute_input":"2024-03-28T15:37:46.666880Z","iopub.status.idle":"2024-03-28T15:37:46.714315Z","shell.execute_reply.started":"2024-03-28T15:37:46.666839Z","shell.execute_reply":"2024-03-28T15:37:46.712714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selection = ['credacc_actualbalance_314A', 'credacc_maxhisbal_375A', 'credacc_minhisbal_90A', 'credacc_status_367L', 'credacc_transactions_402L', 'isdebitcard_527L', 'revolvingaccount_394A']\ntest_applprev_1 = test_applprev_1.drop(selection)\n\nselection = ['annualeffectiverate_199L', 'annualeffectiverate_63L', 'contractsum_5085717L', 'credlmt_230A', 'credlmt_935A', 'debtoutstand_525A', 'debtoverdue_47A', 'instlamount_768A', 'instlamount_852A', 'interestrate_508L', 'nominalrate_281L', 'numberofcontrsvalue_258L', 'numberofcontrsvalue_358L', 'numberofinstls_320L', 'numberofoutstandinstls_59L', 'numberofoverdueinstlmaxdat_641D', 'outstandingamount_362A', 'overdueamountmax2date_1142D', 'periodicityofpmts_837L', 'prolongationcount_1120L', 'prolongationcount_599L', 'residualamount_488A', 'residualamount_856A', 'totalamount_996A', 'totaldebtoverduevalue_178A', 'totaldebtoverduevalue_718A', 'totaloutstanddebtvalue_39A', 'totaloutstanddebtvalue_668A']\ntest_credit_bureau_a_1 = test_credit_bureau_a_1.drop(selection)\n\n# Change L columns to float64\nfor col in test_credit_bureau_a_1.columns:\n        if col[-1] in (\"L\"):\n            test_credit_bureau_a_1 = test_credit_bureau_a_1.with_columns(pl.col(col).cast(pl.Float64).alias(col))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.715965Z","iopub.execute_input":"2024-03-28T15:37:46.716420Z","iopub.status.idle":"2024-03-28T15:37:46.728312Z","shell.execute_reply.started":"2024-03-28T15:37:46.716377Z","shell.execute_reply":"2024-03-28T15:37:46.726997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_applprev_1_feats_1 = test_applprev_1.group_by(\"case_id\").agg(\n    pl.col(\"actualdpd_943P\").sum().alias(\"actualdpd_943P_sum\"),\n    pl.col(\"annuity_853A\").sum().alias(\"annuity_853A_sum\"),\n    pl.col(\"byoccupationinc_3656910L\").max().alias(\"byoccupationinc_3656910L_max\"),\n    pl.col(\"childnum_21L\").max().alias(\"childnum_21L_max\"),\n    pl.col(\"credacc_credlmt_575A\").max().alias(\"credacc_credlmt_575A_max\"),\n    pl.col(\"currdebt_94A\").sum().alias(\"currdebt_94A_sum\"),\n    pl.col(\"downpmt_134A\").sum().alias(\"downpmt_134A_sum\"),\n    pl.col(\"isbidproduct_390L\").max(),\n    pl.col(\"mainoccupationinc_437A\").sum().alias(\"mainoccupationinc_437A_sum\"),\n    pl.col(\"maxdpdtolerance_577P\").max().alias(\"maxdpdtolerance_577P_max\"),\n    pl.col(\"outstandingdebt_522A\").sum().alias(\"outstandingdebt_522A_sum\"),\n    pl.col(\"pmtnum_8L\").sum().alias(\"pmtnum_8L_sum\"),\n    pl.col(\"tenor_203L\").sum().alias(\"tenor_203L_sum\"))\n\ntest_applprev_1_feats_2 = test_applprev_1.select([\"case_id\", \"num_group1\",\n    \"credtype_587L\",\"familystate_726L\",\"inittransactioncode_279L\",\"status_219L\"]).filter(\n    pl.col(\"num_group1\")==0).drop(\"num_group1\")\n\ntest_tax_registry_a_1_feats = test_tax_registry_a_1.group_by(\"case_id\").agg(\n    pl.col(\"amount_4527230A\").sum().alias(\"amount_4527230A_sum\"))\n\ntest_tax_registry_b_1_feats = test_tax_registry_b_1.group_by(\"case_id\").agg(\n    pl.col(\"amount_4917619A\").sum().alias(\"amount_4917619A_sum\"))\n\ntest_tax_registry_c_1_feats = test_tax_registry_c_1.group_by(\"case_id\").agg(\n    pl.col(\"pmtamount_36A\").sum().alias(\"pmtamount_36A_sum\"))\n\ntest_credit_bureau_a_1_feats = test_credit_bureau_a_1.group_by(\"case_id\").agg(\n    pl.col(\"dpdmax_139P\").max().alias(\"dpdmax_139P_max\"),\n    pl.col(\"dpdmax_757P\").max().alias(\"dpdmax_757P_max\"),\n    pl.col(\"monthlyinstlamount_332A\").sum().alias(\"monthlyinstlamount_332A_sum\"),\n    pl.col(\"monthlyinstlamount_674A\").sum().alias(\"monthlyinstlamount_674A_sum\"),\n    pl.col(\"nominalrate_498L\").max().alias(\"nominalrate_498L_max\"),\n    pl.col(\"numberofinstls_229L\").sum().alias(\"numberofinstls_229L_sum\"),\n    pl.col(\"numberofoutstandinstls_520L\").sum().alias(\"numberofoutstandinstls_520L_sum\"),\n    pl.col(\"numberofoverdueinstlmax_1039L\").sum().alias(\"numberofoverdueinstlmax_1039L_sum\"),\n    pl.col(\"numberofoverdueinstlmax_1151L\").sum().alias(\"numberofoverdueinstlmax_1151L_sum\"),\n    pl.col(\"numberofoverdueinstls_725L\").sum().alias(\"numberofoverdueinstls_725L_sum\"),\n    pl.col(\"numberofoverdueinstls_834L\").sum().alias(\"numberofoverdueinstls_834L_sum\"),\n    pl.col(\"outstandingamount_354A\").sum().alias(\"outstandingamount_354A_sum\"),\n    pl.col(\"overdueamount_31A\").sum().alias(\"overdueamount_31A_sum\"),\n    pl.col(\"overdueamount_659A\").sum().alias(\"overdueamount_659A_sum\"),\n    pl.col(\"overdueamountmax2_14A\").max().alias(\"overdueamountmax2_14A_max\"),\n    pl.col(\"overdueamountmax2_398A\").max().alias(\"overdueamountmax2_398A_max\"),\n    pl.col(\"overdueamountmax_155A\").max().alias(\"overdueamountmax_155A_max\"),\n    pl.col(\"overdueamountmax_35A\").max().alias(\"overdueamountmax_35A_max\"),\n    pl.col(\"periodicityofpmts_1102L\").max().alias(\"periodicityofpmts_1102L_max\"),\n    pl.col(\"totalamount_6A\").sum().alias(\"totalamount_6A_sum\"))\n\ntest_tax_registry_c_1_feats= test_tax_registry_c_1_feats.with_columns(pl.col('case_id').cast(pl.Int64))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.729903Z","iopub.execute_input":"2024-03-28T15:37:46.730591Z","iopub.status.idle":"2024-03-28T15:37:46.758696Z","shell.execute_reply.started":"2024-03-28T15:37:46.730556Z","shell.execute_reply":"2024-03-28T15:37:46.757179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"join_data2 = test_basetable.join(test_applprev_1_feats_1, how=\"left\", on=\"case_id\"\n).join(test_applprev_1_feats_2, how=\"left\", on=\"case_id\"\n).join(test_tax_registry_a_1_feats, how=\"left\", on=\"case_id\"\n).join(test_tax_registry_b_1_feats, how=\"left\", on=\"case_id\"\n).join(test_tax_registry_c_1_feats, how=\"left\", on=\"case_id\"\n).join(test_credit_bureau_a_1_feats, how=\"left\", on=\"case_id\")\n\njoin_data2=join_data2.to_pandas()\n\ndrop_cols = ['date_decision','MONTH','WEEK_NUM','byoccupationinc_3656910L_max','familystate_726L', 'amount_4527230A_sum', 'amount_4917619A_sum', 'pmtamount_36A_sum']\njoin_data2 = join_data2.drop(drop_cols, axis=1, errors='ignore')\njoin_data2.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.768520Z","iopub.execute_input":"2024-03-28T15:37:46.768927Z","iopub.status.idle":"2024-03-28T15:37:46.790651Z","shell.execute_reply.started":"2024-03-28T15:37:46.768896Z","shell.execute_reply":"2024-03-28T15:37:46.789193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_applprev_1, test_tax_registry_a_1, test_tax_registry_b_1, test_tax_registry_c_1,test_credit_bureau_a_1\ndel test_applprev_1_feats_1, test_applprev_1_feats_2, test_tax_registry_a_1_feats, test_tax_registry_b_1_feats, test_tax_registry_c_1_feats, test_credit_bureau_a_1_feats\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.792377Z","iopub.execute_input":"2024-03-28T15:37:46.793058Z","iopub.status.idle":"2024-03-28T15:37:46.924787Z","shell.execute_reply.started":"2024-03-28T15:37:46.793020Z","shell.execute_reply":"2024-03-28T15:37:46.923420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Part 3","metadata":{}},{"cell_type":"code","source":"# Additional testing data, depth = 2\ntest_applprev_2 = pl.read_csv(pathway + \"csv_files/test/test_applprev_2.csv\").pipe(set_table_dtypes)\n\ntest_person_2 = pl.read_csv(pathway + \"csv_files/test/test_person_2.csv\").pipe(set_table_dtypes)\n\nsel = ['case_id','num_group1','num_group2','pmts_dpd_1073P','pmts_dpd_303P','pmts_overdue_1140A','pmts_overdue_1152A']\ntest_credit_bureau_a_2 = pl.concat(\n    [pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_0.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_1.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_2.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_3.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_4.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_5.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_6.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_7.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_8.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_9.csv\",columns=sel).pipe(set_table_dtypes),\n    pl.read_csv(pathway + \"csv_files/test/test_credit_bureau_a_2_10.csv\",columns=sel).pipe(set_table_dtypes)\n    ], how=\"vertical_relaxed\")","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.926443Z","iopub.execute_input":"2024-03-28T15:37:46.926805Z","iopub.status.idle":"2024-03-28T15:37:46.974500Z","shell.execute_reply.started":"2024-03-28T15:37:46.926776Z","shell.execute_reply":"2024-03-28T15:37:46.973490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_applprev_2_feats = test_applprev_2.select([\"case_id\", \"num_group1\", \"num_group2\",\n    \"conts_type_509L\"]).filter(\n    (pl.col(\"num_group1\")==0) & (pl.col(\"num_group2\")==0)).drop(\"num_group1\").drop(\"num_group2\") \n\ntest_credit_bureau_a_2_feats = test_credit_bureau_a_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_dpd_1073P\").sum().alias(\"pmts_dpd_1073P_sum\"),\n    pl.col(\"pmts_dpd_303P\").sum().alias(\"pmts_dpd_303P_sum\"),\n    pl.col(\"pmts_overdue_1140A\").sum().alias(\"pmts_overdue_1140A_sum\"),\n    pl.col(\"pmts_overdue_1152A\").sum().alias(\"pmts_overdue_1152A_sum\"))\n\ntest_person_2_feats = test_person_2.select([\"case_id\", \"num_group1\", \"num_group2\", \"addres_zip_823M\",\n    \"addres_district_368M\",\"conts_role_79M\",\"empls_economicalst_849M\",\"empls_employer_name_740M\"]).filter(\n    (pl.col(\"num_group1\")==0) & (pl.col(\"num_group2\")==0)).drop(\"num_group1\").drop(\"num_group2\")","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.976040Z","iopub.execute_input":"2024-03-28T15:37:46.976771Z","iopub.status.idle":"2024-03-28T15:37:46.989990Z","shell.execute_reply.started":"2024-03-28T15:37:46.976737Z","shell.execute_reply":"2024-03-28T15:37:46.988572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"join_data3 = test_basetable.join(test_applprev_2_feats, how=\"left\", on=\"case_id\"\n).join(test_credit_bureau_a_2_feats, how=\"left\", on=\"case_id\").join(test_person_2_feats, how=\"left\", on=\"case_id\")\n\njoin_data3 = join_data3.to_pandas()\njoin_data3 = join_data3.drop(['date_decision','MONTH','WEEK_NUM'], axis=1, errors='ignore')","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:46.992270Z","iopub.execute_input":"2024-03-28T15:37:46.992823Z","iopub.status.idle":"2024-03-28T15:37:47.015053Z","shell.execute_reply.started":"2024-03-28T15:37:46.992773Z","shell.execute_reply":"2024-03-28T15:37:47.013817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_applprev_2, test_person_2, test_credit_bureau_a_2\ndel test_applprev_2_feats, test_credit_bureau_a_2_feats, test_person_2_feats\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.017228Z","iopub.execute_input":"2024-03-28T15:37:47.017700Z","iopub.status.idle":"2024-03-28T15:37:47.147640Z","shell.execute_reply.started":"2024-03-28T15:37:47.017652Z","shell.execute_reply":"2024-03-28T15:37:47.146058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = [join_data1, join_data2, join_data3]\njoin_test = reduce(lambda left, right: pd.merge(left, right, on='case_id'), dfs)\n\n# Convert back to polars for datetime \njoin_test = pl.from_pandas(join_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.149420Z","iopub.execute_input":"2024-03-28T15:37:47.149996Z","iopub.status.idle":"2024-03-28T15:37:47.197854Z","shell.execute_reply.started":"2024-03-28T15:37:47.149960Z","shell.execute_reply":"2024-03-28T15:37:47.196617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final Join Test Data","metadata":{}},{"cell_type":"code","source":"join_test = join_test.with_columns(pl.col('date_decision','birth_259D').cast(pl.Date))\n# Feature engineer days difference between decision and birthdate\ntest = join_test.with_columns(\n    ((pl.col(\"date_decision\") - pl.col(\"birth_259D\")) / (24 * 60 * 60 * 1000)).cast(pl.Float64).alias(\"date_diff\"))\n\n# Drop uneeded date + other columns\ndate_list = ['date_decision','MONTH','firstdatedue_489D','lastactivateddate_801D','lastapplicationdate_877D', 'lastapprdate_640D', 'dateofbirth_337D', 'firstclxcampaign_1125D', \n             'birth_259D','datefirstoffer_1144D', 'datelastunpaid_3546854D', 'lastrejectdate_50D', 'maxdpdinstldate_3546855D', 'responsedate_1012D', 'responsedate_4527233D', 'requesttype_4525192L']\n\ntest = test.pipe(set_table_dtypes).pipe(convert_strings)\n\n# Convert to pandas for drop(errors='ignore')\ntest = test.to_pandas()\ntest = test.drop(date_list, axis=1, errors='ignore')\n\ndel join_data1, join_data2, join_data3, join_test\ngc.collect()\n\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.199978Z","iopub.execute_input":"2024-03-28T15:37:47.200447Z","iopub.status.idle":"2024-03-28T15:37:47.373488Z","shell.execute_reply.started":"2024-03-28T15:37:47.200409Z","shell.execute_reply":"2024-03-28T15:37:47.372175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Noticed some of these numeric variables were parsed as strings, changed them all to int \nnum_list = ['days120_123L','days180_256L','days30_165L','days360_512L','days90_310L','firstquarter_103L','numinstpaidlastcontr_4325080L',\n            'fourthquarter_440L','numinstlswithdpd5_4187116L','numberofqueries_373L', 'secondquarter_766L','thirdquarter_1082L']\nfor col in num_list:\n    test[col]=test[col].astype('float64')\n\n# Drop more unneeded columns, all based on training missing values, 0 range for numeric, or one unique category\ndrop_list = ['lastapprcommoditytypec_5251766M', 'lastrejectcommodtypec_5251769M','lastrejectcommoditycat_161M','lastrejectreasonclient_4145040M',\n            'previouscontdistrict_112M','lastapprcommoditycat_1041M', 'lastcancelreason_561M','lastrejectreason_759M', 'addres_zip_823M', 'addres_district_368M',\n            'commnoinclast6m_3546845L', 'deferredmnthsnum_166L', 'mastercontrelectronic_519L', 'mastercontrexist_109L','paytype1st_925L_OTHER','paytype_783L_OTHER','applicationcnt_361L','empls_employer_name_740M_a55475b1']\ntest.drop(columns=drop_list,axis=1,errors='ignore',inplace=True)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.374713Z","iopub.execute_input":"2024-03-28T15:37:47.375113Z","iopub.status.idle":"2024-03-28T15:37:47.434812Z","shell.execute_reply.started":"2024-03-28T15:37:47.375068Z","shell.execute_reply":"2024-03-28T15:37:47.432645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Imputation","metadata":{}},{"cell_type":"code","source":"print(np.count_nonzero(test.isnull()))\n# Impute missing values, 0 missing values after imputation\ntest = imputer(test)\nprint(np.count_nonzero(test.isnull()))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.436893Z","iopub.execute_input":"2024-03-28T15:37:47.437338Z","iopub.status.idle":"2024-03-28T15:37:47.562156Z","shell.execute_reply.started":"2024-03-28T15:37:47.437303Z","shell.execute_reply":"2024-03-28T15:37:47.560823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See boolean columns\nbool_cols = test.select_dtypes(include=['bool']).columns.tolist()\n# Convert boolean columns to 0 or 1 (False or True)\nfor col in bool_cols:\n    test[col] = test[col].astype(int)\n# Check unique values of bool_cols\nfor col in bool_cols:\n    print(test[col].unique())","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.563796Z","iopub.execute_input":"2024-03-28T15:37:47.564980Z","iopub.status.idle":"2024-03-28T15:37:47.576930Z","shell.execute_reply.started":"2024-03-28T15:37:47.564922Z","shell.execute_reply":"2024-03-28T15:37:47.575629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = test.select_dtypes(include=['category']).columns.tolist()\n# Create dummies for all cat columns, not dropping first to keep column names same as training\ntest = pd.get_dummies(test, dtype=int, columns=cat_cols, sparse=True, drop_first=False)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.578761Z","iopub.execute_input":"2024-03-28T15:37:47.579212Z","iopub.status.idle":"2024-03-28T15:37:47.660125Z","shell.execute_reply.started":"2024-03-28T15:37:47.579178Z","shell.execute_reply":"2024-03-28T15:37:47.658920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Model","metadata":{}},{"cell_type":"code","source":"train = pl.read_csv('/kaggle/input/training/train_final_dummy.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:47.661501Z","iopub.execute_input":"2024-03-28T15:37:47.661841Z","iopub.status.idle":"2024-03-28T15:37:57.911655Z","shell.execute_reply.started":"2024-03-28T15:37:47.661812Z","shell.execute_reply":"2024-03-28T15:37:57.910289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert polars df to pandas so pandas specific methods/attributes work later, seems more memory efficient to load as pl and convert to pd than load as pd\ntrain = train.to_pandas()\n# Get list of ids for submission file\nids = test['case_id'].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:37:57.912812Z","iopub.execute_input":"2024-03-28T15:37:57.913290Z","iopub.status.idle":"2024-03-28T15:38:01.564905Z","shell.execute_reply.started":"2024-03-28T15:37:57.913258Z","shell.execute_reply":"2024-03-28T15:38:01.563619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Only select common columns to use\ncommon_columns = list(set(train.columns) & set(test.columns))\n\ntest=test[common_columns]\n\n# Subset train with only columns seen in test + target\ntrain = train[common_columns+['target']]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:01.566420Z","iopub.execute_input":"2024-03-28T15:38:01.566769Z","iopub.status.idle":"2024-03-28T15:38:02.481500Z","shell.execute_reply.started":"2024-03-28T15:38:01.566739Z","shell.execute_reply":"2024-03-28T15:38:02.480084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# Fit on stratified sample\n# Note: no random seed, warning message is fine\ntrain_sample = train.groupby('target', group_keys=False).apply(lambda x: x.sample(frac=0.01)).reset_index(drop=True)\ntrain_sample.head()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:02.484344Z","iopub.execute_input":"2024-03-28T15:38:02.484749Z","iopub.status.idle":"2024-03-28T15:38:04.246840Z","shell.execute_reply.started":"2024-03-28T15:38:02.484715Z","shell.execute_reply":"2024-03-28T15:38:04.245450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train.loc[:,'target'].to_frame('target')\nX = train.drop(['target',], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.248524Z","iopub.execute_input":"2024-03-28T15:38:04.249717Z","iopub.status.idle":"2024-03-28T15:38:04.261876Z","shell.execute_reply.started":"2024-03-28T15:38:04.249680Z","shell.execute_reply":"2024-03-28T15:38:04.260751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check target distribution is same after all the preprocessing/sampling\nprint(round(y.target.value_counts()[1]/y.target.value_counts().sum(),4))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.263509Z","iopub.execute_input":"2024-03-28T15:38:04.264200Z","iopub.status.idle":"2024-03-28T15:38:04.272347Z","shell.execute_reply.started":"2024-03-28T15:38:04.264165Z","shell.execute_reply":"2024-03-28T15:38:04.271187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.274275Z","iopub.execute_input":"2024-03-28T15:38:04.274793Z","iopub.status.idle":"2024-03-28T15:38:04.412574Z","shell.execute_reply.started":"2024-03-28T15:38:04.274749Z","shell.execute_reply":"2024-03-28T15:38:04.411043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Do not include case_id, or week_num as numeric \nnumeric_cols = test.select_dtypes(include=['number']).columns.tolist()\nnumeric_cols.remove('case_id')\nnumeric_cols.remove('WEEK_NUM')\n# print(numeric_cols)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.414366Z","iopub.execute_input":"2024-03-28T15:38:04.415966Z","iopub.status.idle":"2024-03-28T15:38:04.425758Z","shell.execute_reply.started":"2024-03-28T15:38:04.415923Z","shell.execute_reply":"2024-03-28T15:38:04.424634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nscaler = MinMaxScaler(copy=False)\nX[numeric_cols] = scaler.fit_transform(X[numeric_cols])\ntest[numeric_cols] = scaler.transform(test[numeric_cols])\nwarnings.filterwarnings(\"default\")\n\nX.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.427138Z","iopub.execute_input":"2024-03-28T15:38:04.428120Z","iopub.status.idle":"2024-03-28T15:38:04.582600Z","shell.execute_reply.started":"2024-03-28T15:38:04.428058Z","shell.execute_reply":"2024-03-28T15:38:04.581333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 75/25 split\n# X_train, X_valid, y_train, y_valid= train_test_split(X, y, test_size=0.25, stratify=y, random_state = 123)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.583844Z","iopub.execute_input":"2024-03-28T15:38:04.584211Z","iopub.status.idle":"2024-03-28T15:38:04.588776Z","shell.execute_reply.started":"2024-03-28T15:38:04.584181Z","shell.execute_reply":"2024-03-28T15:38:04.587687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del X, y\n#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.590538Z","iopub.execute_input":"2024-03-28T15:38:04.591268Z","iopub.status.idle":"2024-03-28T15:38:04.599466Z","shell.execute_reply.started":"2024-03-28T15:38:04.591235Z","shell.execute_reply":"2024-03-28T15:38:04.597959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check proper split\n#print(X_train.shape)\n#print(X_valid.shape)\n#print(y_train.shape)\n#print(y_valid.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.601048Z","iopub.execute_input":"2024-03-28T15:38:04.601459Z","iopub.status.idle":"2024-03-28T15:38:04.611910Z","shell.execute_reply.started":"2024-03-28T15:38:04.601429Z","shell.execute_reply":"2024-03-28T15:38:04.610537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SMOTE/NEARMISS HERE","metadata":{}},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\nprint(\"Before Undersampling, counts of label '1': {}\".format(sum(y_train['target'] == 1))) \nprint(\"Before Undersampling, counts of label '0': {} \\n\".format(sum(y_train['target'] == 0)))\n\n# Version 1: Default, selects samples of the majority class for which average distances to the k closest instances of the minority class is smallest\n# Version 2: Selects samples of the majority class for which average distances to the k farthest instances of the minority class is smallest.\n# Version 3: Firstly, for each minority class instance, their M nearest-neighbors will be stored. Then finally, the majority class instances are selected for which the average distance to the N nearest-neighbors is the largest.\nnr = NearMiss(sampling_strategy=0.25)\n\n\nX_train_miss, y_train_miss = nr.fit_resample(X_train, y_train)\nprint('After Undersampling, the shape of train_X: {}'.format(X_train_miss.shape)) \nprint('After Undersampling, the shape of train_y: {} \\n'.format(y_train_miss.shape)) \n  \nprint(\"After Undersampling, counts of label '1': {}\".format(sum(y_train_miss['target'] == 1))) \nprint(\"After Undersampling, counts of label '0': {}\".format(sum(y_train_miss['target'] == 0)))\nwarnings.filterwarnings(\"default\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.615004Z","iopub.execute_input":"2024-03-28T15:38:04.615521Z","iopub.status.idle":"2024-03-28T15:38:04.626676Z","shell.execute_reply.started":"2024-03-28T15:38:04.615487Z","shell.execute_reply":"2024-03-28T15:38:04.625191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\nprint(\"Before Oversampling, counts of label '1': {}\".format(sum(y_train['target'] == 1))) \nprint(\"Before Oversampling, counts of label '0': {} \\n\".format(sum(y_train['target'] == 0)))\n\nsmote = SMOTE(sampling_strategy = 0.25, random_state=123)\n\nX_train_smote, y_train_smote = smote.fit_resample(X_train, y_train)\nprint('After Oversampling, the shape of train_X: {}'.format(X_train_smote.shape)) \nprint('After Oversampling, the shape of train_y: {} \\n'.format(y_train_smote.shape)) \n  \nprint(\"After Oversampling, counts of label '1': {}\".format(sum(y_train_smote['target'] == 1))) \nprint(\"After Oversampling, counts of label '0': {}\".format(sum(y_train_smote['target'] == 0))) \nwarnings.filterwarnings(\"default\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.630294Z","iopub.execute_input":"2024-03-28T15:38:04.630805Z","iopub.status.idle":"2024-03-28T15:38:04.641993Z","shell.execute_reply.started":"2024-03-28T15:38:04.630760Z","shell.execute_reply":"2024-03-28T15:38:04.641032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models","metadata":{}},{"cell_type":"code","source":"\"\"\"\n# The “balanced” mode uses the values of y to automatically adjust weights inversely proportional \n# to class frequencies in the input data as n_samples / (n_classes * np.bincount(y)).\nclf = LogisticRegression(class_weight='balanced', random_state=123)\ngrid_params = {\n    'solver': ['lbfgs','newton-cg','newton-cholesky','sag'], # these 4 solvers either use l2 or None penalty\n    'penalty': ['l2', None], 'C': [0.01, 0.1, 1, 10], # Regularization parameter, default = 1.0\n}\n\n# Using roc_auc instead of default accuracy to score due to imbalanced target\n# Competition evaluation metric is based off AUC\ngrid_search = GridSearchCV(clf, grid_params, verbose=1, scoring='roc_auc', cv=3)\nprint(\"Hyperparameters to tune are:\")\npprint(grid_params)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.643065Z","iopub.execute_input":"2024-03-28T15:38:04.643465Z","iopub.status.idle":"2024-03-28T15:38:04.655898Z","shell.execute_reply.started":"2024-03-28T15:38:04.643436Z","shell.execute_reply":"2024-03-28T15:38:04.654663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nclf = RandomForestClassifier(class_weight='balanced', random_state=123)\ngrid_params = {\n    'n_estimators': [100, 250, 500], # default: 100\n    'criterion': ['gini', 'entropy', 'log_loss'], # default: gini\n    #'max_features': ['sqrt', 'log2', None] # default: sqrt\n}\n\n# Using roc_auc instead of default accuracy to score due to imbalanced target\n# Competition evaluation metric is based off AUC\ngrid_search = GridSearchCV(clf, grid_params, verbose=1, scoring='roc_auc', cv=3)\nprint(\"Hyperparameters to tune are:\")\npprint(grid_params)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.665707Z","iopub.execute_input":"2024-03-28T15:38:04.666227Z","iopub.status.idle":"2024-03-28T15:38:04.674654Z","shell.execute_reply.started":"2024-03-28T15:38:04.666187Z","shell.execute_reply":"2024-03-28T15:38:04.673342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop case_id and week_num from features, leave original X_train and X_valid for metric scoring later\nweeks = X[\"WEEK_NUM\"]\nX_feats = X.drop(['case_id', 'WEEK_NUM'], axis=1)\n# X_valid_feats = X_valid.drop(['case_id', 'WEEK_NUM'], axis=1)\n\n# Sort columns in alphabetical order for training so columns match test submission\nX_feats = X_feats.reindex(sorted(X_feats.columns), axis=1)\n# X_valid_feats = X_valid_feats.reindex(sorted(X_valid_feats.columns), axis=1)\n\nprint(X_feats.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.676362Z","iopub.execute_input":"2024-03-28T15:38:04.676703Z","iopub.status.idle":"2024-03-28T15:38:04.724345Z","shell.execute_reply.started":"2024-03-28T15:38:04.676675Z","shell.execute_reply":"2024-03-28T15:38:04.722892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_feats.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.726216Z","iopub.execute_input":"2024-03-28T15:38:04.726569Z","iopub.status.idle":"2024-03-28T15:38:04.766989Z","shell.execute_reply.started":"2024-03-28T15:38:04.726540Z","shell.execute_reply":"2024-03-28T15:38:04.765607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For testing loaded model and manually defining columns to use\n# print(list(X_feats))","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.769183Z","iopub.execute_input":"2024-03-28T15:38:04.769552Z","iopub.status.idle":"2024-03-28T15:38:04.774984Z","shell.execute_reply.started":"2024-03-28T15:38:04.769521Z","shell.execute_reply":"2024-03-28T15:38:04.773661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwarnings.filterwarnings(\"ignore\")\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\nfitted_models = []\ncv_scores = []\n\n# default boosting type: gbdt. note: these params were taken from another competition notebook \ngrid_params = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"class_weight\": 'balanced',\n    \"max_depth\": 10,  \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,  \n    \"colsample_bytree\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 123,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\":True,\n    'num_leaves':64\n}\n\nfor idx_train, idx_valid in cv.split(X_feats, y, groups=weeks):\n    X_train, y_train = X_feats.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X_feats.iloc[idx_valid], y.iloc[idx_valid]\n    \n    clf = lgb.LGBMClassifier(**grid_params)\n    clf.fit(\n        X_train, y_train,\n        eval_set = [(X_valid, y_valid)],\n        callbacks = [lgb.log_evaluation(200), lgb.early_stopping(100)])\n    fitted_models.append(clf)\n    \n    y_pred_valid = clf.predict_proba(X_valid)[:,1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores.append(auc_score)\n\nprint(\"CV AUC scores: \", cv_scores)\nprint(\"Maximum CV AUC score: \", max(cv_scores))\n\nwarnings.filterwarnings(\"default\")","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:04.776500Z","iopub.execute_input":"2024-03-28T15:38:04.776850Z","iopub.status.idle":"2024-03-28T15:38:22.210293Z","shell.execute_reply.started":"2024-03-28T15:38:04.776821Z","shell.execute_reply":"2024-03-28T15:38:22.208765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the best cv model\nmodel = fitted_models[np.argmax(cv_scores)]","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.212071Z","iopub.execute_input":"2024-03-28T15:38:22.212633Z","iopub.status.idle":"2024-03-28T15:38:22.223177Z","shell.execute_reply.started":"2024-03-28T15:38:22.212589Z","shell.execute_reply":"2024-03-28T15:38:22.221730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n%%time\n# 40 min to fit on 5% LR model with grid search params\n# 30 min for 5% RF model (less params), 1 hr for 10%\n# lgb model 30% 3 folds: 38 min to run notebook\nwarnings.filterwarnings(\"ignore\")\n\ngrid_search.fit(X_train_feats, y_train)\n\nwarnings.filterwarnings('default')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.224737Z","iopub.execute_input":"2024-03-28T15:38:22.225213Z","iopub.status.idle":"2024-03-28T15:38:22.234275Z","shell.execute_reply.started":"2024-03-28T15:38:22.225178Z","shell.execute_reply":"2024-03-28T15:38:22.232816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# Only run with grid search\nprint(f\"Best score= {grid_search.best_score_:0.3f}\")\nprint(\"Best parameters set:\")\nbest_parameters = grid_search.best_estimator_.get_params()\nfor name in sorted(grid_params.keys()):\n    print(\"\\t%s: %r\" % (name, best_parameters[name]))\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.235747Z","iopub.execute_input":"2024-03-28T15:38:22.236153Z","iopub.status.idle":"2024-03-28T15:38:22.250804Z","shell.execute_reply.started":"2024-03-28T15:38:22.236112Z","shell.execute_reply":"2024-03-28T15:38:22.249486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\ny_pred = grid_search.predict(X_valid_feats)\n\nprint(f'Validation Target: {round(y_valid.target.value_counts()[1]/y_valid.target.value_counts().sum(),4)}')\nprint(f'Validation Accuracy: {accuracy_score(y_valid,y_pred)}')\nwarnings.filterwarnings('default')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.252142Z","iopub.execute_input":"2024-03-28T15:38:22.252555Z","iopub.status.idle":"2024-03-28T15:38:22.264715Z","shell.execute_reply.started":"2024-03-28T15:38:22.252522Z","shell.execute_reply":"2024-03-28T15:38:22.263365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\n# Print confusion matrix with percent\ntarget_names= ['No Default', 'Default']\nmatrix = confusion_matrix(y_valid, y_pred, normalize='true')\ncm_display = ConfusionMatrixDisplay(confusion_matrix= matrix, display_labels=target_names).plot()\nwarnings.filterwarnings('default')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.266512Z","iopub.execute_input":"2024-03-28T15:38:22.267150Z","iopub.status.idle":"2024-03-28T15:38:22.282124Z","shell.execute_reply.started":"2024-03-28T15:38:22.267083Z","shell.execute_reply":"2024-03-28T15:38:22.280662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\n# Print out the report\nprint(classification_report(y_valid, y_pred, target_names = target_names))\nwarnings.filterwarnings('default')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.283513Z","iopub.execute_input":"2024-03-28T15:38:22.283962Z","iopub.status.idle":"2024-03-28T15:38:22.296610Z","shell.execute_reply.started":"2024-03-28T15:38:22.283922Z","shell.execute_reply":"2024-03-28T15:38:22.295158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# best model params, only run with grid search\nbest_model = grid_search.best_estimator_\nprint(best_model)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.298207Z","iopub.execute_input":"2024-03-28T15:38:22.298675Z","iopub.status.idle":"2024-03-28T15:38:22.309325Z","shell.execute_reply.started":"2024-03-28T15:38:22.298636Z","shell.execute_reply":"2024-03-28T15:38:22.307957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission Notes\n1. Lr model class_weight=None, submission score: 0.415. \n\n2. Lr model class_weight='balanced', submission score: 0.452. Gini function predicted 0.53 on validation. Using 5%, \n\n3. RF model class_weight='balanced', submission score: 0.481 using 10%, score: 0.497 using 15%.\n\n4. RF model using same params as #3 with 50% sample, submission score: 0.484 (prob due to no cv). \n\n5. RF model using same params as #3 with 50% sample but now with cv using 1 param. Submission score: 0.493 with predicted 0.59 on validation. Surprising that 5x the sample did not improve the score. \n\n5. LGBM model using full training and sgkf 5 folds, no class_weight=balanced. Submission score: 0.538.\n\n6. LGBM model with class_weight = balanced. Submission score: 0.491! Was worse\n\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\n# Plot ROC AUC curve using skplt\nwarnings.filterwarnings(\"ignore\")\ny_prob = grid_search.predict_proba(X_feats)\nskplt.metrics.plot_roc_curve(y_valid, y_prob)\nplt.show()\nwarnings.filterwarnings('default')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.311164Z","iopub.execute_input":"2024-03-28T15:38:22.311636Z","iopub.status.idle":"2024-03-28T15:38:22.322820Z","shell.execute_reply.started":"2024-03-28T15:38:22.311596Z","shell.execute_reply":"2024-03-28T15:38:22.321574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# Save model\njoblib.dump(grid_search, 'rf_model15.joblib')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.324795Z","iopub.execute_input":"2024-03-28T15:38:22.325524Z","iopub.status.idle":"2024-03-28T15:38:22.335610Z","shell.execute_reply.started":"2024-03-28T15:38:22.325479Z","shell.execute_reply":"2024-03-28T15:38:22.334397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Metric Scoring","metadata":{}},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\n\nbase_train = pd.concat([X_train, y_train], axis=1)\nbase_train['score'] = grid_search.predict_proba(X_train_feats)[:,1]\nprint(f\"The AUC score on the train set is: {roc_auc_score(base_train['target'], base_train['score'])}\")\n\nbase_valid = pd.concat([X_valid, y_valid], axis=1)\nbase_valid['score'] = grid_search.predict_proba(X_valid_feats)[:,1]\nprint(f\"The AUC score on the valid set is: {roc_auc_score(base_valid['target'], base_valid['score'])}\")\n\nwarnings.filterwarnings('default')\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.336929Z","iopub.execute_input":"2024-03-28T15:38:22.337492Z","iopub.status.idle":"2024-03-28T15:38:22.348336Z","shell.execute_reply.started":"2024-03-28T15:38:22.337460Z","shell.execute_reply":"2024-03-28T15:38:22.347367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Taken from competition starter notebook made by one of the organizers \n# Note: may not work based on how we sampled a low percent of the data because some weeks will be all 0 targets\n# Using 1 and 2% samples did not work\n# Using 5% only worked for train, not for valid\ndef gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.349533Z","iopub.execute_input":"2024-03-28T15:38:22.349941Z","iopub.status.idle":"2024-03-28T15:38:22.361560Z","shell.execute_reply.started":"2024-03-28T15:38:22.349844Z","shell.execute_reply":"2024-03-28T15:38:22.360111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\n# supress lets the notebook run even if there is an exception thrown here from the sampling\nwith suppress(Exception):\n    stability_score_train = gini_stability(base_train)\n    print(f'The stability score on the train set is: {stability_score_train}') \n\nwarnings.filterwarnings(\"default\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.363121Z","iopub.execute_input":"2024-03-28T15:38:22.363670Z","iopub.status.idle":"2024-03-28T15:38:22.382837Z","shell.execute_reply.started":"2024-03-28T15:38:22.363559Z","shell.execute_reply":"2024-03-28T15:38:22.381197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nwarnings.filterwarnings(\"ignore\")\n\nwith suppress(Exception):\n    stability_score_valid = gini_stability(base_valid)\n    print(f'The stability score on the valid set is: {stability_score_valid}') \n\nwarnings.filterwarnings(\"default\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.384321Z","iopub.execute_input":"2024-03-28T15:38:22.384908Z","iopub.status.idle":"2024-03-28T15:38:22.392959Z","shell.execute_reply.started":"2024-03-28T15:38:22.384874Z","shell.execute_reply":"2024-03-28T15:38:22.391681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_feats, X_train, X_valid, y_train, y_valid, X, y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.394587Z","iopub.execute_input":"2024-03-28T15:38:22.395006Z","iopub.status.idle":"2024-03-28T15:38:22.529684Z","shell.execute_reply.started":"2024-03-28T15:38:22.394948Z","shell.execute_reply":"2024-03-28T15:38:22.528267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"cell_type":"code","source":"\"\"\"\n# get importance for linear regression, may need to change grid_search to model? \nimportance = grid_search.coef_[0]\n# summarize feature importance\nfor i,v in enumerate(importance):\n print('Feature: %0d, Score: %.5f' % (i,v))\n# plot feature importance\nplt.bar([x for x in range(len(importance))], importance)\nplt.show()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.531336Z","iopub.execute_input":"2024-03-28T15:38:22.531821Z","iopub.status.idle":"2024-03-28T15:38:22.543467Z","shell.execute_reply.started":"2024-03-28T15:38:22.531782Z","shell.execute_reply":"2024-03-28T15:38:22.542186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# get importance for tree models\nimportance = model.feature_importances_\n# summarize feature importance\nfor i,v in enumerate(importance):\n print('Feature: %0d, Score: %.5f' % (i,v))\n# plot feature importance\nplt.bar([x for x in range(len(importance))], importance)\nplt.show()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.545011Z","iopub.execute_input":"2024-03-28T15:38:22.545524Z","iopub.status.idle":"2024-03-28T15:38:22.561701Z","shell.execute_reply.started":"2024-03-28T15:38:22.545469Z","shell.execute_reply":"2024-03-28T15:38:22.560335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For lgb model\nlgb.plot_importance(model, figsize=(15,45))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:22.563287Z","iopub.execute_input":"2024-03-28T15:38:22.563733Z","iopub.status.idle":"2024-03-28T15:38:25.448466Z","shell.execute_reply.started":"2024-03-28T15:38:22.563660Z","shell.execute_reply":"2024-03-28T15:38:25.447293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\n#### To do: 1. Drop more variables based on feature importance (compare lr, rf, and lgbm)\n#### 2. Possibly incorporate PCA\n\n#### 3. Evaluate and submit other models\n\nstability metric = mean(gini) + 88.0⋅min(0,a) − 0.5⋅std(residuals) where\ngini = 2∗AUC−1","metadata":{}},{"cell_type":"code","source":"# Sort columns alphabetically to match loaded model\ntest = test.reindex(sorted(test.columns), axis=1)\n\npredictions = model.predict_proba(test.drop(['case_id', 'WEEK_NUM'], axis=1))\nprint(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:25.449712Z","iopub.execute_input":"2024-03-28T15:38:25.450651Z","iopub.status.idle":"2024-03-28T15:38:25.480470Z","shell.execute_reply.started":"2024-03-28T15:38:25.450615Z","shell.execute_reply":"2024-03-28T15:38:25.479126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:25.482852Z","iopub.execute_input":"2024-03-28T15:38:25.483348Z","iopub.status.idle":"2024-03-28T15:38:25.643484Z","shell.execute_reply.started":"2024-03-28T15:38:25.483311Z","shell.execute_reply":"2024-03-28T15:38:25.642184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'case_id': ids, 'score': predictions[:,1]}).set_index('case_id')\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:25.644883Z","iopub.execute_input":"2024-03-28T15:38:25.645261Z","iopub.status.idle":"2024-03-28T15:38:25.660394Z","shell.execute_reply.started":"2024-03-28T15:38:25.645228Z","shell.execute_reply":"2024-03-28T15:38:25.659076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-28T15:38:25.662076Z","iopub.execute_input":"2024-03-28T15:38:25.662542Z","iopub.status.idle":"2024-03-28T15:38:25.672316Z","shell.execute_reply.started":"2024-03-28T15:38:25.662508Z","shell.execute_reply":"2024-03-28T15:38:25.671343Z"},"trusted":true},"execution_count":null,"outputs":[]}]}