{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport gc\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport polars as pl\nfrom xgboost import XGBClassifier\ndataPath = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-29T07:30:57.218459Z","iopub.execute_input":"2024-03-29T07:30:57.219286Z","iopub.status.idle":"2024-03-29T07:30:58.138885Z","shell.execute_reply.started":"2024-03-29T07:30:57.219240Z","shell.execute_reply":"2024-03-29T07:30:58.137902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making sklearn pipeline outputs as dataframe:-\nfrom sklearn import set_config;\nset_config(transform_output = \"pandas\");\npd.set_option('display.max_columns', 50);\npd.set_option('display.max_rows', 50);\npd.set_option('display.precision', 3);","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:30:58.140637Z","iopub.execute_input":"2024-03-29T07:30:58.141037Z","iopub.status.idle":"2024-03-29T07:30:58.145950Z","shell.execute_reply.started":"2024-03-29T07:30:58.141010Z","shell.execute_reply":"2024-03-29T07:30:58.145008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float32).alias(col))\n        if col[-1] == \"D\":\n            df = df.with_columns(pl.col(col).str.to_date())\n\n    return df\n\ndef get_credit_bureau_grp_a0():\n    if Path(\"/kaggle/working/train_credit_bureau_a0.csv\").exists():\n        return pl.read_csv(\"/kaggle/working/train_credit_bureau_a0.csv\")\n    for i in range(4):\n        if i==0:\n            df = pl.read_csv(dataPath/Path(f'csv_files/train/train_credit_bureau_a_1_{i}.csv')).pipe(set_table_dtypes)\n            df = df.filter(pl.col(\"num_group1\")==0)\n        else:\n            df_tmp = pl.read_csv(dataPath/Path(f'csv_files/train/train_credit_bureau_a_1_{i}.csv')).pipe(set_table_dtypes)\n            df = df.filter(pl.col(\"num_group1\")==0)\n            df = pl.concat([df, df_tmp],\n                          how=\"vertical_relaxed\")\n        \n    print(df.shape)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:30:58.147099Z","iopub.execute_input":"2024-03-29T07:30:58.147446Z","iopub.status.idle":"2024-03-29T07:30:58.161583Z","shell.execute_reply.started":"2024-03-29T07:30:58.147414Z","shell.execute_reply":"2024-03-29T07:30:58.160684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_credit_bureau_a0 = get_credit_bureau_grp_a0()\ntrain_credit_bureau_a0.write_csv(\"/kaggle/working/train_credit_bureau_a0.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:30:58.164029Z","iopub.execute_input":"2024-03-29T07:30:58.164740Z","iopub.status.idle":"2024-03-29T07:31:15.945450Z","shell.execute_reply.started":"2024-03-29T07:30:58.164715Z","shell.execute_reply":"2024-03-29T07:31:15.944465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_credit_bureau_b0 = pl.write_csv(dataPath/f'csv_files/train/train_credit_bureau_b_1.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:15.946633Z","iopub.execute_input":"2024-03-29T07:31:15.946949Z","iopub.status.idle":"2024-03-29T07:31:15.951535Z","shell.execute_reply.started":"2024-03-29T07:31:15.946924Z","shell.execute_reply":"2024-03-29T07:31:15.950362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    \n    @staticmethod\n    def filter_dense_col(df):\n        cols_drop = []\n        for col in df.columns:\n            if col not in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                null_ratio = df[col].is_null().mean()\n                if null_ratio > 0.95:\n                    cols_drop.append(col)\n        print(f'dropping columns: {cols_drop} because they are not dense')\n        df = df.drop(columns=cols_drop)\n        return df\n    \n    @staticmethod\n    def filter_based_on_mode(df):\n        cols_drop = []\n        for col in df.columns:\n            if col not in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"] and col[-1] not in [\"P\", \"A\", \"D\"]:\n                cnt_unique = df[col].n_unique()\n                if cnt_unique==1 or cnt_unique>100:\n                    cols_drop.append(col)\n        print(f'dropping columns: {cols_drop} because of count')\n        df = df.drop(columns=cols_drop)\n        return df\n                    \n    @staticmethod\n    def get_weeks_from_decision(df):\n        date_cols = []\n        for col in df.columns:\n            if col[-1]==\"D\" and df[col].dtype!=pl.String:\n                print(col)\n                try:\n                    df = df.with_columns(pl.col(\"date_decision\").sub(pl.col(col)).dt.total_days().alias(f'diff_{col}'))\n                    df = df.with_columns([pl.col(f'diff_{col}') // 7])\n                    df = df.with_columns(pl.col(f'diff_{col}').cast(pl.Int16))\n                except Exception as e:\n                    print(f'{col} \\t \\t {str(e)}')\n                else:\n                    date_cols.append(col)\n                \n        df = df.drop(columns=date_cols)\n        return df\n","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:15.952807Z","iopub.execute_input":"2024-03-29T07:31:15.953150Z","iopub.status.idle":"2024-03-29T07:31:15.965504Z","shell.execute_reply.started":"2024-03-29T07:31:15.953118Z","shell.execute_reply":"2024-03-29T07:31:15.964593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_base = pl.read_csv(dataPath/Path(\"csv_files/train/train_base.csv\")).pipe(set_table_dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:15.967636Z","iopub.execute_input":"2024-03-29T07:31:15.968019Z","iopub.status.idle":"2024-03-29T07:31:16.120182Z","shell.execute_reply.started":"2024-03-29T07:31:15.967996Z","shell.execute_reply":"2024-03-29T07:31:16.119233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # convert info to dict for easy access\n# df_info = pl.read_csv(dataPath/\"feature_definitions.csv\")\n# df_info = df_info.set_index(\"Variable\")[\"Description\"].to_dict()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:16.121515Z","iopub.execute_input":"2024-03-29T07:31:16.122034Z","iopub.status.idle":"2024-03-29T07:31:16.126467Z","shell.execute_reply.started":"2024-03-29T07:31:16.121998Z","shell.execute_reply":"2024-03-29T07:31:16.125434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ntrain_base[\"WEEK_NUM\"].nunique() = 92\ntrain_base[\"target\"].value_counts()\n0    1478665\n1      47994\n47994/len(train_base) = 0.0314\nlen(train_base) = 1526659\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:16.127829Z","iopub.execute_input":"2024-03-29T07:31:16.128107Z","iopub.status.idle":"2024-03-29T07:31:16.139730Z","shell.execute_reply.started":"2024-03-29T07:31:16.128084Z","shell.execute_reply":"2024-03-29T07:31:16.138798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### lets start adding one data source at a time and create features, train a base model, and get some metrics","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_static_0 = pl.read_csv(dataPath/Path(\"csv_files/train/train_static_0_0.csv\")).pipe(set_table_dtypes)\ntrain_static_1 = pl.read_csv(dataPath/Path(\"csv_files/train/train_static_0_1.csv\")).pipe(set_table_dtypes)\n\ntrain_static = pl.concat([train_static_0, train_static_1], how=\"vertical_relaxed\")\ndel [train_static_0, train_static_1]\n\n# train_static_0.shape = (1003757, 168)\ntrain_static.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:16.140889Z","iopub.execute_input":"2024-03-29T07:31:16.141151Z","iopub.status.idle":"2024-03-29T07:31:26.029461Z","shell.execute_reply.started":"2024-03-29T07:31:16.141129Z","shell.execute_reply":"2024-03-29T07:31:26.028634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### data1 - debit card data\n- select case with group_num1 = 0 for now\n- drop Debit card opening date.","metadata":{}},{"cell_type":"code","source":"%%time\n# lets try to see debit card data\ntrain_debit = pl.read_csv(dataPath/Path(\"csv_files/train/train_debitcard_1.csv\")).pipe(set_table_dtypes)\n\nlen(train_debit)\n# (157302, 111772)\n\ntrain_debit = train_debit.filter(pl.col(\"num_group1\")==0).drop(columns=[\"openingdate_857D\"])\ntrain_debit = train_debit.drop_nulls(subset=[\"last180dayaveragebalance_704A\", \"last180dayturnover_1134A\",\"last30dayturnover_651A\"])\nprint(train_debit.shape)\ntrain_debit.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:26.030455Z","iopub.execute_input":"2024-03-29T07:31:26.030726Z","iopub.status.idle":"2024-03-29T07:31:26.111013Z","shell.execute_reply.started":"2024-03-29T07:31:26.030702Z","shell.execute_reply":"2024-03-29T07:31:26.110166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_info[\"last180dayaveragebalance_704A\"], df_info[\"last180dayturnover_1134A\"], df_info[\"last30dayturnover_651A\"], df_info[\"openingdate_857D\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:26.111978Z","iopub.execute_input":"2024-03-29T07:31:26.112267Z","iopub.status.idle":"2024-03-29T07:31:26.116425Z","shell.execute_reply.started":"2024-03-29T07:31:26.112243Z","shell.execute_reply":"2024-03-29T07:31:26.115586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### now let's look at train_static_cb_0","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_static_cb_0 = pl.read_csv(dataPath/Path(\"csv_files/train/train_static_cb_0.csv\")).pipe(set_table_dtypes)\ntrain_static_cb_0.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:26.117686Z","iopub.execute_input":"2024-03-29T07:31:26.118005Z","iopub.status.idle":"2024-03-29T07:31:28.605528Z","shell.execute_reply.started":"2024-03-29T07:31:26.117979Z","shell.execute_reply":"2024-03-29T07:31:28.604645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # get na ratio of various columns . using this we will select cols that seem useful\n# nan_ratio = train_static_cb_0.isnull().mean().to_dict()\n# for col, na_ratio_value in nan_ratio.items():\n#     if na_ratio_value<0.8:\n#         if col in df_info:\n#             print(f'{col}\\t:--->\\t {df_info[col]}\\t {round(na_ratio_value,2)}')\n#         else:\n#             print(f'{col}\\t:--->\\t NA \\t {round(na_ratio_value,2)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:28.606623Z","iopub.execute_input":"2024-03-29T07:31:28.606904Z","iopub.status.idle":"2024-03-29T07:31:28.610931Z","shell.execute_reply.started":"2024-03-29T07:31:28.606879Z","shell.execute_reply":"2024-03-29T07:31:28.610068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### merge all the tables","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df = train_static.join(train_debit, on=[\"case_id\"], how=\"left\")\ntrain_df = train_df.join(train_static_cb_0, on=[\"case_id\"], how=\"left\")\ntrain_df = train_df.join(train_credit_bureau_a0, on=[\"case_id\"], how=\"left\")\ntrain_df = train_df.join(train_base, on=[\"case_id\"], how=\"left\")","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:28.614728Z","iopub.execute_input":"2024-03-29T07:31:28.615054Z","iopub.status.idle":"2024-03-29T07:31:35.314093Z","shell.execute_reply.started":"2024-03-29T07:31:28.615030Z","shell.execute_reply":"2024-03-29T07:31:35.312930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ctypes;\nlibc = ctypes.CDLL(\"libc.so.6\");\nfrom gc import collect\nfrom os import path, walk, getpid;\nfrom psutil import Process;","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:35.315111Z","iopub.execute_input":"2024-03-29T07:31:35.315367Z","iopub.status.idle":"2024-03-29T07:31:35.320017Z","shell.execute_reply.started":"2024-03-29T07:31:35.315345Z","shell.execute_reply":"2024-03-29T07:31:35.319120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_memory():\n    \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\";\n\n    collect();\n    libc.malloc_trim(0);\n    pid        = getpid();\n    py         = Process(pid);\n    memory_use = py.memory_info()[0] / 2. ** 30;\n    return f\"\\nRAM usage = {memory_use :.4} GB\";","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:35.321220Z","iopub.execute_input":"2024-03-29T07:31:35.321548Z","iopub.status.idle":"2024-03-29T07:31:35.332631Z","shell.execute_reply.started":"2024-03-29T07:31:35.321518Z","shell.execute_reply":"2024-03-29T07:31:35.331703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_memory()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:35.333833Z","iopub.execute_input":"2024-03-29T07:31:35.334456Z","iopub.status.idle":"2024-03-29T07:31:35.407613Z","shell.execute_reply.started":"2024-03-29T07:31:35.334425Z","shell.execute_reply":"2024-03-29T07:31:35.406780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_list = [train_static, train_debit, train_static_cb_0, train_credit_bureau_a0, train_base]\ndel del_list\nclean_memory()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:35.408744Z","iopub.execute_input":"2024-03-29T07:31:35.409819Z","iopub.status.idle":"2024-03-29T07:31:35.463620Z","shell.execute_reply.started":"2024-03-29T07:31:35.409787Z","shell.execute_reply":"2024-03-29T07:31:35.462706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = train_df.with_columns(pl.col(\"date_decision\").str.to_date())","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:35.464916Z","iopub.execute_input":"2024-03-29T07:31:35.465561Z","iopub.status.idle":"2024-03-29T07:31:36.471108Z","shell.execute_reply.started":"2024-03-29T07:31:35.465528Z","shell.execute_reply":"2024-03-29T07:31:36.470150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = Pipeline.get_weeks_from_decision(train_df)\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:36.472657Z","iopub.execute_input":"2024-03-29T07:31:36.473068Z","iopub.status.idle":"2024-03-29T07:31:41.749153Z","shell.execute_reply.started":"2024-03-29T07:31:36.473033Z","shell.execute_reply":"2024-03-29T07:31:41.748243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = Pipeline.filter_dense_col(train_df)\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:41.750329Z","iopub.execute_input":"2024-03-29T07:31:41.750615Z","iopub.status.idle":"2024-03-29T07:31:43.544943Z","shell.execute_reply.started":"2024-03-29T07:31:41.750591Z","shell.execute_reply":"2024-03-29T07:31:43.544034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = Pipeline.filter_based_on_mode(train_df)\nprint(train_df.shape)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:43.545982Z","iopub.execute_input":"2024-03-29T07:31:43.546276Z","iopub.status.idle":"2024-03-29T07:31:48.724979Z","shell.execute_reply.started":"2024-03-29T07:31:43.546243Z","shell.execute_reply":"2024-03-29T07:31:48.723920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_memory()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:48.726244Z","iopub.execute_input":"2024-03-29T07:31:48.726560Z","iopub.status.idle":"2024-03-29T07:31:48.797279Z","shell.execute_reply.started":"2024-03-29T07:31:48.726535Z","shell.execute_reply":"2024-03-29T07:31:48.796347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:31:49.132297Z","iopub.execute_input":"2024-03-29T07:31:49.132662Z","iopub.status.idle":"2024-03-29T07:31:54.795539Z","shell.execute_reply.started":"2024-03-29T07:31:49.132630Z","shell.execute_reply":"2024-03-29T07:31:54.794449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_memory()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:32:27.281857Z","iopub.execute_input":"2024-03-29T07:32:27.282231Z","iopub.status.idle":"2024-03-29T07:32:27.414594Z","shell.execute_reply.started":"2024-03-29T07:32:27.282203Z","shell.execute_reply":"2024-03-29T07:32:27.413635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.memory_usage().sum()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:32:30.145136Z","iopub.execute_input":"2024-03-29T07:32:30.145836Z","iopub.status.idle":"2024-03-29T07:32:30.162205Z","shell.execute_reply.started":"2024-03-29T07:32:30.145793Z","shell.execute_reply":"2024-03-29T07:32:30.161268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.memory_usage().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:32:45.391416Z","iopub.execute_input":"2024-03-29T07:32:45.392124Z","iopub.status.idle":"2024-03-29T07:32:45.403067Z","shell.execute_reply.started":"2024-03-29T07:32:45.392090Z","shell.execute_reply":"2024-03-29T07:32:45.402157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"diff_responsedate_4917613D\"].mean().abs()","metadata":{"execution":{"iopub.status.busy":"2024-03-29T07:33:50.906582Z","iopub.execute_input":"2024-03-29T07:33:50.907332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.write_csv(\"/kaggle/working/train.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%reset -f","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_csv(\"/kaggle/working/train.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_to_category(df):\n    for col in df.columns:\n        if df[col].dtype.str==\"|O\":\n            df[col] = df[col].astype('category')\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nfrom xgboost import XGBClassifier","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = convert_to_category(train_df)\n\n# create train and val df\nreal_train_df , val_df = train_test_split(train_df)\nreal_train_df.shape, val_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_train_df = real_train_df.drop(columns=[\"case_id\", \"MONTH\", \"WEEK_NUM\", \"target\"])\nxgb_val_df = val_df.drop(columns=[\"case_id\", \"MONTH\", \"WEEK_NUM\", \"target\"])\nxgb_train_label = real_train_df[[\"target\"]]\nxgb_val_label = val_df[[\"target\"]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del [real_train_df]\nclean_memory()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del [train_df]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_memory()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"diff_columns = [x for x in xgb_train_df.columns if x.startswith(\"diff\") and x[-1]==\"D\"]\nlen(diff_columns)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in diff_columns:\n    xgb_train_df[col] = xgb_train_df[col].fillna(-1)\n    xgb_train_df[col] = xgb_train_df[col].astype('int8')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_memory()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nxgb_train_df.replace([np.inf, -np.inf], np.nan, inplace=True)\nxgb_val_df.replace([np.inf, -np.inf], np.nan, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in xgb_train_df.columns:\n    if xgb_train_df[x].dtype==float:\n        xgb_train_df[x] = xgb_train_df[x].fillna(0)\n        xgb_train_df[x] = xgb_train_df[x].astype(np.int16)\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_memory()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_train_df.memory_usage().sort_values(ascending=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nxgb_train_df.to_csv(\"/kaggle/working/train_final.csv\", index=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nxgb_val_df.to_csv(\"/kaggle/working/val_final.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nxgb_train_label.to_csv(\"/kaggle/working/train_label.csv\")\nxgb_val_label.to_csv(\"/kaggle/working/val_label.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%reset -f","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport polars as pl\nfrom xgboost import XGBClassifier\nimport pandas as pd","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit\nxgb_train_df = pd.read_csv(\"/kaggle/working/train_final.csv\")\nxgb_val_df = pd.read_csv(\"/kaggle/working/val_final.csv\")\nxgb_train_label = pd.read_csv(\"/kaggle/working/train_label.csv\")\nxgb_val_df = pd.read_csv(\"/kaggle/working/val_label.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bst = XGBClassifier(n_estimators=10, max_depth=4,objective='binary:logistic', enable_categorical=True, scale_pos_weight=15)\n# fit model\nbst.fit(xgb_train_df, xgb_train_label)\n# make predictions\npreds = bst.predict_proba(xgb_val_df)\nroc_auc_score(xgb_val_label, preds[:,1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open(\"/kaggle/working/model.pkl\", \"wb\") as model_path:\n    pickle.dump(bst, model_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols = xgb_train_df.columns\nwith open(\"/kaggle/working/train_cols.pkl\", \"wb\") as cols_path:\n    pickle.dump(train_cols, cols_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in train_cols:\n    if x[:4]==\"diff\":\n        print(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%reset -f","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Keeping note of various iterations\n- **ver 1**: just use train static, convert all object dtypes to category and train xgboost model with n_estimators=3, max_depth=3. got val auc of 0.6927\n\n- **ver 2**: just use train static, convert all object dtypes to category and train xgboost model with n_estimators=4, max_depth=6. got val auc of 0.72\n\n- **ver 3**: use debit card, cb data, select columns with nan ratio < 0.5, n_estimators=50, got val auc of 0.74","metadata":{}},{"cell_type":"markdown","source":"## make predictions on test set\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport gc\nimport pandas as pd\n\nfrom pathlib import Path\nimport pickle\ndataPath = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_base = pd.read_csv(dataPath/Path(\"csv_files/test/test_base.csv\"))\ntest_base.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_static_0 = pd.read_csv(dataPath/Path(\"csv_files/test/test_static_0_0.csv\"))\ntest_static_1 = pd.read_csv(dataPath/Path(\"csv_files/test/test_static_0_1.csv\"))\ntest_static_2 = pd.read_csv(dataPath/Path(\"csv_files/test/test_static_0_2.csv\"))\n\ntest_static = pd.concat([test_static_0, test_static_1, test_static_2])\ndel test_static_0, test_static_1, test_static_2\ntest_feat = test_base.merge(test_static, on=[\"case_id\"], how=\"left\")\ndel test_static, test_base\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_debit = pd.read_csv(dataPath/Path(\"csv_files/test/test_debitcard_1.csv\"))\ntest_debit = test_debit[test_debit[\"num_group1\"]==0].drop(columns=[\"openingdate_857D\"])\ntest_debit = test_debit.dropna(subset=[\"last180dayaveragebalance_704A\", \"last180dayturnover_1134A\",\"last30dayturnover_651A\"], how=\"all\")\n\ntest_feat = test_feat.merge(test_debit, on=[\"case_id\"], how=\"left\")\ndel test_debit\ntest_static_cb_0 = pd.read_csv(dataPath/Path(\"csv_files/test/test_static_cb_0.csv\"))\ntest_feat = test_feat.merge(test_static_cb_0, on=[\"case_id\"], how=\"left\")\ndel test_static_cb_0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_credit_bureau_grp_a0_test():\n    for i in range(5):\n        if i==0:\n            df = pd.read_csv(dataPath/Path(f'csv_files/test/test_credit_bureau_a_1_{i}.csv'))\n            df = df[df[\"num_group1\"]==0]\n        else:\n            df_tmp = pd.read_csv(dataPath/Path(f'csv_files/test/test_credit_bureau_a_1_{i}.csv'))\n            df_tmp = df_tmp[df_tmp[\"num_group1\"]==0]\n            df = pd.concat([df, df_tmp])\n\n        print(df.shape)\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_credit_bureau_grp_a0 = get_credit_bureau_grp_a0_test()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_feat = test_feat.merge(test_credit_bureau_grp_a0, on=[\"case_id\"], how=\"left\")\ndel test_credit_bureau_grp_a0\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/working/model.pkl\", \"rb\") as model_path:\n    bst = pickle.load(model_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/working/train_cols.pkl\", \"rb\") as cols_path:\n    train_cols = pickle.load(cols_path)\nxgb_test = test_feat[train_cols]\nprint(xgb_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_dtype(df):\n    for col in df.columns:\n        if df[col].dtype.str == \"|O\":\n            df[col] = df[col].fillna(\"\")\n            df[col] = df[col].astype('category')\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_test = convert_dtype(xgb_test)\npreds_test = bst.predict_proba(xgb_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(dataPath / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = preds_test[:,1]\ndf_subm.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}