{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30648,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\n# !pip install \nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nimport pickle as pkl\nimport  gc\nimport glob\nfrom tqdm import tqdm \n# !pip install pyspark\n# import pyspark\n# from pyspark.sql import SparkSession\n# spark = SparkSession.builder.appName(\"CreditRiskModeling\").getOrCreate()\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         if 'train' in filename:\n#             print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-13T17:41:38.192537Z","iopub.execute_input":"2024-02-13T17:41:38.193845Z","iopub.status.idle":"2024-02-13T17:41:41.228791Z","shell.execute_reply.started":"2024-02-13T17:41:38.193783Z","shell.execute_reply":"2024-02-13T17:41:41.227803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install varclushi\n!pip install optbinning","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:41:41.230693Z","iopub.execute_input":"2024-02-13T17:41:41.231006Z","iopub.status.idle":"2024-02-13T17:42:16.798834Z","shell.execute_reply.started":"2024-02-13T17:41:41.230980Z","shell.execute_reply":"2024-02-13T17:42:16.797719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from optbinning import OptimalBinning","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:16.800851Z","iopub.execute_input":"2024-02-13T17:42:16.801316Z","iopub.status.idle":"2024-02-13T17:42:17.547828Z","shell.execute_reply.started":"2024-02-13T17:42:16.801272Z","shell.execute_reply":"2024-02-13T17:42:17.546889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EXPLORING BASE DATA","metadata":{}},{"cell_type":"code","source":"#Helper functions\ndef reduce_mem_usage(df, int_cast=True, obj_to_category=False, subset=None):\n    \"\"\"\n    Iterate through all the columns of a dataframe and modify the data type to reduce memory usage.\n    :param df: dataframe to reduce (pd.DataFrame)\n    :param int_cast: indicate if columns should be tried to be casted to int (bool)\n    :param obj_to_category: convert non-datetime related objects to category dtype (bool)\n    :param subset: subset of columns to analyse (list)\n    :return: dataset with the column dtypes adjusted (pd.DataFrame)\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024 ** 2;\n    gc.collect()\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n#     cols_none = subset if subset is  None else df.columns.tolist()\n#     for col_non in tqdm(cols_none):\n#         df[col_non] = df[col_non].fillna(-888)\n    \n    cols = subset if subset is not None else df.columns.tolist()\n\n    for col in tqdm(cols):\n        col_type = df[col].dtype\n\n        if col_type != object and col_type.name != 'category' and 'datetime' not in col_type.name:\n            df[col] = df[col].fillna(-888)\n            c_min = df[col].min()\n            c_max = df[col].max()\n\n#             # test if column can be converted to an integer\n#             treat_as_int = str(col_type)[:3] == 'int'\n#             if int_cast and not treat_as_int:\n#                 treat_as_int = check_if_integer(df[col])\n                \n            treat_as_int = True\n            if treat_as_int:\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.uint8).min and c_max < np.iinfo(np.uint8).max:\n                    df[col] = df[col].astype(np.uint8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.uint16).min and c_max < np.iinfo(np.uint16).max:\n                    df[col] = df[col].astype(np.uint16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.uint32).min and c_max < np.iinfo(np.uint32).max:\n                    df[col] = df[col].astype(np.uint32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n                elif c_min > np.iinfo(np.uint64).min and c_max < np.iinfo(np.uint64).max:\n                    df[col] = df[col].astype(np.uint64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        elif 'datetime' not in col_type.name and obj_to_category:\n            df[col] = df[col].fillna('Mis')\n            df[col] = df[col].astype('category')\n    gc.collect()\n    end_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage after optimization is: {:.3f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n\ndef date_column_depth_0(df):\n    date_columns = ['date_decision'] + [x for x in df.columns if x[-1] == 'D'] \n    df[date_columns] = df[date_columns].apply(pd.to_datetime, errors='coerce')\n    df_diff = df[date_columns].apply(lambda col: (df['date_decision'] - col).dt.days)\n    df_diff.columns = [f'Diff_{col}' for col in df_diff.columns]\n    df = pd.concat([df, df_diff], axis=1)\n    return df\n\n\ndef union_parquest(list_parq):\n    df_list = [reduce_mem_usage(pd.read_parquet(i)) for i in list_parq]\n    union_df = pd.concat(df_list)\n    union_df = reduce_mem_usage(union_df)\n    return union_df   \n\ndef gini(x):\n    total = 0\n    for i, xi in enumerate(x[:-1], 1):\n        total += np.sum(np.abs(xi - x[i:]))\n    return total / (len(x)**2 * np.mean(x))","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:17.550551Z","iopub.execute_input":"2024-02-13T17:42:17.551437Z","iopub.status.idle":"2024-02-13T17:42:17.581214Z","shell.execute_reply.started":"2024-02-13T17:42:17.551398Z","shell.execute_reply":"2024-02-13T17:42:17.579766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def multi_merge(base_data,train_vs_test,data_type):\n    if train_vs_test ==  'train':\n        file_path = train_files_path\n        list_parq =  [file_path + '/' + i for i in os.listdir(file_path) if data_type in i ] \n        \n    elif train_vs_test ==  'test':\n        file_path = test_files_path\n        list_parq =  [file_path + '/' + i for i in os.listdir(file_path) if data_type in i ] \n        \n    df_i_merged = pd.DataFrame()\n    \n    for i in list_parq:\n        print(i)\n        df_i = pd.read_parquet(i)\n        df_i = reduce_mem_usage(df_i)\n        if 'num_group1' in df_i.columns: \n            df_i = df_i[df_i['num_group1'] == 0 ]\n            df_i = df_i.drop(columns = 'num_group1')\n    #         df_i_merged = df_i_merged.merge(df_i,how = 'left',on = 'case_id')\n        df_i_merged = pd.concat([df_i_merged,df_i])\n        del df_i\n        gc.collect()\n    return df_i_merged        ","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:17.582485Z","iopub.execute_input":"2024-02-13T17:42:17.582804Z","iopub.status.idle":"2024-02-13T17:42:17.600034Z","shell.execute_reply.started":"2024-02-13T17:42:17.582778Z","shell.execute_reply":"2024-02-13T17:42:17.599027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files_path = '/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/'\ntest_files_path = '/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/'","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:17.601576Z","iopub.execute_input":"2024-02-13T17:42:17.602032Z","iopub.status.idle":"2024-02-13T17:42:17.616898Z","shell.execute_reply.started":"2024-02-13T17:42:17.601995Z","shell.execute_reply":"2024-02-13T17:42:17.615894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Base Data","metadata":{}},{"cell_type":"code","source":"#Train\ntrain_base_df = pd.read_parquet(train_files_path + 'train_base.parquet')\ntrain_base_df = reduce_mem_usage(train_base_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:17.618253Z","iopub.execute_input":"2024-02-13T17:42:17.618701Z","iopub.status.idle":"2024-02-13T17:42:18.369630Z","shell.execute_reply.started":"2024-02-13T17:42:17.618664Z","shell.execute_reply":"2024-02-13T17:42:18.368448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_pivot = pd.pivot_table(train_base_df,index = ['WEEK_NUM','MONTH'],aggfunc= {'case_id':'count','target':'sum'}).reset_index()\nbase_pivot['perc_bad'] = base_pivot['target']/base_pivot['case_id'] ","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:18.371180Z","iopub.execute_input":"2024-02-13T17:42:18.372022Z","iopub.status.idle":"2024-02-13T17:42:18.530850Z","shell.execute_reply.started":"2024-02-13T17:42:18.371974Z","shell.execute_reply":"2024-02-13T17:42:18.529819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_pivot","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:18.532194Z","iopub.execute_input":"2024-02-13T17:42:18.532659Z","iopub.status.idle":"2024-02-13T17:42:18.554157Z","shell.execute_reply.started":"2024-02-13T17:42:18.532620Z","shell.execute_reply":"2024-02-13T17:42:18.552970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> BAD PERCENT(MOST IMPORTANT METRIC WEEK-WISE)","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nsns.barplot(base_pivot,x = 'WEEK_NUM',y = 'perc_bad')\nplt.xticks(rotation = 90,size = 7)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:18.557728Z","iopub.execute_input":"2024-02-13T17:42:18.558054Z","iopub.status.idle":"2024-02-13T17:42:20.236432Z","shell.execute_reply.started":"2024-02-13T17:42:18.558027Z","shell.execute_reply":"2024-02-13T17:42:20.235601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see from above plot bad percent is quite high between week 52-63)- this corrosponds to covid times, below chart with month will confirm it\n","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nsns.barplot(base_pivot,x = 'MONTH',y = 'perc_bad')\nplt.xticks(rotation = 90,size = 7)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:20.237495Z","iopub.execute_input":"2024-02-13T17:42:20.237989Z","iopub.status.idle":"2024-02-13T17:42:21.225549Z","shell.execute_reply.started":"2024-02-13T17:42:20.237961Z","shell.execute_reply":"2024-02-13T17:42:21.224323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"WE will have to either remove or treat these special months so that model stability is kept in longer runs.","metadata":{}},{"cell_type":"code","source":"base_pivot[base_pivot['perc_bad'] > base_pivot['perc_bad'].mean()*1.5]","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:21.227443Z","iopub.execute_input":"2024-02-13T17:42:21.227895Z","iopub.status.idle":"2024-02-13T17:42:21.243119Z","shell.execute_reply.started":"2024-02-13T17:42:21.227854Z","shell.execute_reply":"2024-02-13T17:42:21.241936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Static DEPTH 0","metadata":{}},{"cell_type":"code","source":"#For this purpose we will load individual file types and do EDA(for modelling purpose all files were loaded at same time)\n# Load train static data\ndf_merged = train_base_df[['case_id']]\nvariable_type_list = ['train_static_0']\n\nfor k in variable_type_list:\n    df_k = multi_merge(train_base_df,'train',k)\n    df_merged = df_merged.merge(df_k,how = 'outer',on = 'case_id')\n    del df_k\n    gc.collect()\n    \n    \n#Merge with Base\ntrain_static_0 = train_base_df.merge(df_merged,how = 'left',on = 'case_id')\n\n#Fill Missialue\nnum_cols = train_static_0.select_dtypes(include=np.number).columns\ntrain_static_0[num_cols] = train_static_0[num_cols].fillna(-888)\n\nobject_cols = train_static_0.select_dtypes(include='object').columns\ntrain_static_0[object_cols] = train_static_0[object_cols].fillna('Mis')","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:42:21.244790Z","iopub.execute_input":"2024-02-13T17:42:21.245235Z","iopub.status.idle":"2024-02-13T17:42:54.080495Z","shell.execute_reply.started":"2024-02-13T17:42:21.245197Z","shell.execute_reply":"2024-02-13T17:42:54.079493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Very few columsn have more than 98% missing values","metadata":{}},{"cell_type":"markdown","source":"Object columns also have few columns less than 99%","metadata":{}},{"cell_type":"code","source":"#Handle Date columns\ndate_columns_train_static_0 = [x for x in train_static_0.columns if x[-1] == 'D']\ntrain_static_0 = date_column_depth_0(train_static_0)\ntrain_static_0 = train_static_0.drop(columns = date_columns_train_static_0)\n\n\n#Fill Missialue\nnum_cols = train_static_0.select_dtypes(include=np.number).columns\ntrain_static_0[num_cols] = train_static_0[num_cols].fillna(-888)\n\nobject_cols = train_static_0.select_dtypes(include='object').columns\ntrain_static_0[object_cols] = train_static_0[object_cols].fillna('Mis')","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:52:56.169045Z","iopub.execute_input":"2024-02-13T17:52:56.169640Z","iopub.status.idle":"2024-02-13T17:53:04.638123Z","shell.execute_reply.started":"2024-02-13T17:52:56.169599Z","shell.execute_reply":"2024-02-13T17:53:04.637103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iv_table = pd.DataFrame()\nfor i in train_static_0.columns[5:]:\n    if np.issubdtype(train_static_0[i].dtype, np.number):\n        optb = OptimalBinning(name=i, dtype=\"numerical\", solver=\"cp\",special_codes = [-888])\n        optb.fit(train_static_0[i].values, train_static_0['target'].values)\n        binning_table = optb.binning_table.build()\n        binning_table['Variable'] = i\n        binning_table = binning_table.drop(index = 'Totals')\n        iv_table = pd.concat([iv_table,binning_table])\n    else:\n        optb = OptimalBinning(name=i, dtype=\"categorical\", solver=\"mip\",special_codes = ['Mis'])\n        optb.fit(train_static_0[i].values, train_static_0['target'].values)\n        binning_table = optb.binning_table.build()\n        binning_table['Variable'] = i\n        binning_table = binning_table.drop(index = 'Totals')\n        iv_table = pd.concat([iv_table,binning_table])\n        print(i)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:53:09.877542Z","iopub.execute_input":"2024-02-13T17:53:09.877995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iv_table","metadata":{"execution":{"iopub.status.busy":"2024-02-13T17:50:47.755912Z","iopub.execute_input":"2024-02-13T17:50:47.757342Z","iopub.status.idle":"2024-02-13T17:50:47.786710Z","shell.execute_reply.started":"2024-02-13T17:50:47.757291Z","shell.execute_reply":"2024-02-13T17:50:47.785700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged = train_base_df[['case_id']]\nvariable_type_list = ['train_static_0',\n                      'train_static_cb_0',\n                      'train_applprev_1',\n                      'train_credit_bureau_a_1',\n                     'train_credit_bureau_b_1',\n                     'train_debitcard_1',\n                     'train_deposit_1',\n                     'train_person_1',\n                     'train_tax_registry_a_1',\n                     'train_tax_registry_b_1',\n                     'train_tax_registry_c_1']\nfor k in variable_type_list:\n    df_k = multi_merge(train_base_df,'train',k)\n    df_merged = df_merged.merge(df_k,how = 'outer',on = 'case_id')\n    del df_k\n    gc.collect()\n    \n    \n#Merge with Base\ndf_merged_train = train_base_df.merge(df_merged,how = 'left',on = 'case_id')\ndel df_merged\n#Convert date columns to difference\ndate_columns_train = [x for x in df_merged_train.columns if x[-1] == 'D']\ndf_merged_train = date_column_depth_0(df_merged_train)\n\n\ndf_merged_train = df_merged_train.drop(columns = date_columns_train)\ngc.collect()\n\n\n#Fill Missialue\nnum_cols = df_merged_train.select_dtypes(include=np.number).columns\ndf_merged_train[num_cols] = df_merged_train[num_cols].fillna(-888)\n\nobject_cols = df_merged_train.select_dtypes(include='object').columns\ndf_merged_train[object_cols] = df_merged_train[object_cols].fillna('Mis')\n\ndf_merged_train = df_merged_train.drop_duplicates(subset= 'case_id')    \n    \n#Reindexing\nidentifier_cols = ['date_decision','MONTH']\ntarget = 'target'\n# Reindex\ndf_merged_train = df_merged_train.set_index(['case_id','WEEK_NUM']) \n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-12T04:50:27.801232Z","iopub.execute_input":"2024-02-12T04:50:27.801521Z","iopub.status.idle":"2024-02-12T04:53:22.325981Z","shell.execute_reply.started":"2024-02-12T04:50:27.801499Z","shell.execute_reply":"2024-02-12T04:53:22.324766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-02-12T04:56:30.871088Z","iopub.execute_input":"2024-02-12T04:56:30.872249Z","iopub.status.idle":"2024-02-12T04:56:31.043065Z","shell.execute_reply.started":"2024-02-12T04:56:30.8722Z","shell.execute_reply":"2024-02-12T04:56:31.041732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-12T04:54:46.183901Z","iopub.execute_input":"2024-02-12T04:54:46.184796Z","iopub.status.idle":"2024-02-12T04:54:46.19025Z","shell.execute_reply.started":"2024-02-12T04:54:46.184767Z","shell.execute_reply":"2024-02-12T04:54:46.189179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define X,y\nX = df_merged_train.drop(columns = identifier_cols + [target])\nX = X.select_dtypes(exclude=['object'])\ny = df_merged_train['target']\n#Delete data\ndel df_merged_train\ngc.collect()\n#Pick some weeks from starting and some weeks from end as OOT\nX_oot = X[X.index.isin([0,  1,  2,  3, \n                        48, 49, 50, 51, 52,\n                        87, 88, 89,90, 91],level = 1)]\ny_oot = y[y.index.isin([0,  1,  2,  3, 48, 49,\n                        50, 51, 52,87, \n                        88, 89,90, 91],level = 1)]\n\n\nX = X[~X.index.isin([0,  1,  2,  3,\n                     48, 49, 50, 51, 52,\n                     87, 88, 89,90, 91],level = 1)]\ny = y[~y.index.isin([0,  1,  2,  3, \n                     48, 49, 50, 51, 52,\n                     87, 88, 89,90, 91],level = 1)]\n\n\n#Train test split(stratified with WEEK_NUM in index 1)\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.25, stratify= list(X.index.get_level_values(1)) , random_state=42)\nX_val, X_test, y_val, y_test = train_test_split(X_val, y_val,stratify= list(X_val.index.get_level_values(1)) ,test_size=0.50, random_state=42)\n#delete\ndel X,y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T04:57:30.862954Z","iopub.execute_input":"2024-02-12T04:57:30.863304Z","iopub.status.idle":"2024-02-12T04:57:38.969254Z","shell.execute_reply.started":"2024-02-12T04:57:30.863277Z","shell.execute_reply":"2024-02-12T04:57:38.968404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params= {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"learning_rate\": 0.5,\n    \"n_estimators\": 1000,\n    \"colsample_bytree\": 0.4, \n    \"colsample_bynode\": 0.4,\n#     \"verbose\": 1,\n    \"random_state\": 42,\n    \"device\": \"cpu\",\n    \"early_stopping_round\": 100\n}\n\nmodel = lgb.LGBMClassifier(**params)\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_val, y_val)])","metadata":{"execution":{"iopub.status.busy":"2024-02-12T04:58:06.934164Z","iopub.execute_input":"2024-02-12T04:58:06.934497Z","iopub.status.idle":"2024-02-12T05:01:04.494366Z","shell.execute_reply.started":"2024-02-12T04:58:06.934472Z","shell.execute_reply":"2024-02-12T05:01:04.493441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import xgboost as xgb\n\n\n# model = xgb.XGBClassifier(\n\n# eta = 0.5,\n# subsample=0.4,\n# eval_metric=\"auc\",\n# colsample_bytree=0.4,\n# min_child_weight=0.4,\n# max_depth=3,\n# learning_rate=0.3,\n# n_estimators=1000,\n# verbose = 2,      \n# device = 'gpu',\n# eval_set=[(X_val, y_val)])\n\n# model.fit(X_train,y_train)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-12T03:26:09.754873Z","iopub.execute_input":"2024-02-12T03:26:09.755162Z","iopub.status.idle":"2024-02-12T03:26:09.759697Z","shell.execute_reply.started":"2024-02-12T03:26:09.755136Z","shell.execute_reply":"2024-02-12T03:26:09.758713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_imp = pd.DataFrame([model.feature_name_,model.feature_importances_],index= ['F','FI']).T","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:02:51.205974Z","iopub.execute_input":"2024-02-12T05:02:51.206329Z","iopub.status.idle":"2024-02-12T05:02:51.220327Z","shell.execute_reply.started":"2024-02-12T05:02:51.206305Z","shell.execute_reply":"2024-02-12T05:02:51.219398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_imp.sort_values('FI',ascending= False).head(50)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:02:52.764581Z","iopub.execute_input":"2024-02-12T05:02:52.765009Z","iopub.status.idle":"2024-02-12T05:02:52.780553Z","shell.execute_reply.started":"2024-02-12T05:02:52.764963Z","shell.execute_reply":"2024-02-12T05:02:52.77953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open('/kaggle/working/base_line_lgbm.pkl', 'wb') as fp:\n#     pkl.dump(model, fp)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:03:57.576079Z","iopub.execute_input":"2024-02-12T05:03:57.57642Z","iopub.status.idle":"2024-02-12T05:03:57.579687Z","shell.execute_reply.started":"2024-02-12T05:03:57.576393Z","shell.execute_reply":"2024-02-12T05:03:57.579045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_pred = model.predict_proba(X_train)[:,1] \ny_val_pred = model.predict_proba(X_val)[:,1]\ny_test_pred = model.predict_proba(X_test)[:,1]\ny_oot_pred = model.predict_proba(X_oot)[:,1]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:03:57.705998Z","iopub.execute_input":"2024-02-12T05:03:57.706611Z","iopub.status.idle":"2024-02-12T05:04:16.889037Z","shell.execute_reply.started":"2024-02-12T05:03:57.706584Z","shell.execute_reply":"2024-02-12T05:04:16.887429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\ndef predict_df(X,y):\n    preds = model.predict_proba(X)[:,1] \n    pred_df = pd.DataFrame(preds,columns = ['predict_proba'],index = X.index)\n    \n# #     pred_df.merge(y,how = 'left',on)\n# #     pred_df['Deciles'] = pd.qcut(train_pred['predict_proba'],q=10,labels = False)  \n#     pred_df = pred_df.set_index(['case_id','WEEK_NUM'])\n    pred_df = pred_df.merge(y,how= 'left',left_index = True,right_index = True) \n    pred_df = pred_df.reset_index(level = 1)\n    return pred_df\n\n\n\n\ndef gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n    ","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:04:30.078693Z","iopub.execute_input":"2024-02-12T05:04:30.079043Z","iopub.status.idle":"2024-02-12T05:04:30.089327Z","shell.execute_reply.started":"2024-02-12T05:04:30.079017Z","shell.execute_reply":"2024-02-12T05:04:30.088013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gini Stability","metadata":{}},{"cell_type":"code","source":"#Train\ntrain_predict_df = predict_df(X_train,y_train)\ntrain_gini_stability = gini_stability(train_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)\n\n\n#Val\nval_predict_df = predict_df(X_val,y_val)\nval_gini_stability = gini_stability(val_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)\n\n#Test\ntest_predict_df = predict_df(X_test,y_test)\ntest_gini_stability = gini_stability(test_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)\n\n#Oot\noot_predict_df = predict_df(X_oot,y_oot)\noot_gini_stability = gini_stability(oot_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:04:31.578093Z","iopub.execute_input":"2024-02-12T05:04:31.578421Z","iopub.status.idle":"2024-02-12T05:04:51.819126Z","shell.execute_reply.started":"2024-02-12T05:04:31.578397Z","shell.execute_reply":"2024-02-12T05:04:51.817835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_gini_stability)\nprint(val_gini_stability)\nprint(test_gini_stability)\nprint(oot_gini_stability)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:04:51.820701Z","iopub.execute_input":"2024-02-12T05:04:51.821011Z","iopub.status.idle":"2024-02-12T05:04:51.825737Z","shell.execute_reply.started":"2024-02-12T05:04:51.82097Z","shell.execute_reply":"2024-02-12T05:04:51.824628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_train = roc_auc_score(y_train,y_train_pred)\nroc_auc_val = roc_auc_score(y_val,y_val_pred)\nroc_auc_test = roc_auc_score(y_test,y_test_pred)\nroc_auc_oot = roc_auc_score(y_oot,y_oot_pred)\n\n# Ginni\nginni_train = roc_auc_train * 2 - 1\nginni_val = roc_auc_val * 2 - 1\nginni_test = roc_auc_test * 2 - 1\nginni_oot = roc_auc_oot * 2 -1","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:12:28.646871Z","iopub.execute_input":"2024-02-12T05:12:28.647423Z","iopub.status.idle":"2024-02-12T05:12:29.1135Z","shell.execute_reply.started":"2024-02-12T05:12:28.647396Z","shell.execute_reply":"2024-02-12T05:12:29.112772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(roc_auc_train)\nprint(roc_auc_val)\nprint(roc_auc_test)\nprint(roc_auc_oot)\n\nprint(ginni_train)\nprint(ginni_val)\nprint(ginni_test)\nprint(ginni_oot)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:12:36.517471Z","iopub.execute_input":"2024-02-12T05:12:36.51806Z","iopub.status.idle":"2024-02-12T05:12:36.523652Z","shell.execute_reply.started":"2024-02-12T05:12:36.518033Z","shell.execute_reply":"2024-02-12T05:12:36.522872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ginni cofficient and stability of model across weeks","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_columns = model.feature_name_","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:12:42.057132Z","iopub.execute_input":"2024-02-12T05:12:42.057714Z","iopub.status.idle":"2024-02-12T05:12:42.063047Z","shell.execute_reply.started":"2024-02-12T05:12:42.057686Z","shell.execute_reply":"2024-02-12T05:12:42.061836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train,X_val,X_test,y_train,y_val,y_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:12:42.252568Z","iopub.execute_input":"2024-02-12T05:12:42.252907Z","iopub.status.idle":"2024-02-12T05:12:43.196291Z","shell.execute_reply.started":"2024-02-12T05:12:42.252882Z","shell.execute_reply":"2024-02-12T05:12:43.195098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %who","metadata":{"execution":{"iopub.status.busy":"2024-02-12T03:26:31.314833Z","iopub.execute_input":"2024-02-12T03:26:31.315169Z","iopub.status.idle":"2024-02-12T03:26:31.324479Z","shell.execute_reply.started":"2024-02-12T03:26:31.315136Z","shell.execute_reply":"2024-02-12T03:26:31.323747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test\ntest_base_df = pd.read_parquet(test_files_path + 'test_base.parquet')\ntest_base_df = reduce_mem_usage(test_base_df)\n\ndf_merged = test_base_df[['case_id']]\nvariable_type_list = ['test_static_0',\n                      'test_static_cb_0',\n                      'test_applprev_1',\n                      'test_credit_bureau_a_1',\n                     'test_credit_bureau_b_1',\n                     'test_debitcard_1',\n                     'test_deposit_1',\n                     'test_person_1',\n                     'test_tax_registry_a_1',\n                     'test_tax_registry_b_1',\n                     'test_tax_registry_c_1']\n\nfor k in variable_type_list:\n    df_k = multi_merge(test_base_df,'test',k)\n    df_merged = df_merged.merge(df_k,how = 'outer',on = 'case_id')\n    del df_k\n    gc.collect()\n    \n    \n#Merge with Base\ndf_merged_test = test_base_df.merge(df_merged,how = 'left',on = 'case_id')\ndel df_merged\n#Convert date columns to difference\ndate_columns_test = [x for x in df_merged_test.columns if x[-1] == 'D']\ndf_merged_test = date_column_depth_0(df_merged_test)\n\n\ndf_merged_test = df_merged_test.drop(columns = date_columns_test)\ngc.collect()\n\n\n#Fill Missialue\nnum_cols = df_merged_test.select_dtypes(include=np.number).columns\ndf_merged_test[num_cols] = df_merged_test[num_cols].fillna(-888)\n\ndf_merged_test['pmtamount_36A'] = df_merged_test['pmtamount_36A'].fillna(-888)\n\nobject_cols = df_merged_test.select_dtypes(include='object').columns\ndf_merged_test[object_cols] = df_merged_test[object_cols].fillna('Mis')\ndf_merged_test = df_merged_test.drop_duplicates(subset= 'case_id')    \n\n    \n    \n#Reindexing\nidentifier_cols = ['date_decision','MONTH','WEEK_NUM']\n#Reindex\ndf_merged_test = df_merged_test.set_index('case_id') \n\n\n\n#Define X,y\ndf_merged_test = df_merged_test.drop(columns = identifier_cols)\ndf_merged_test = df_merged_test[model_columns]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:13:14.888527Z","iopub.execute_input":"2024-02-12T05:13:14.888933Z","iopub.status.idle":"2024-02-12T05:13:38.775459Z","shell.execute_reply.started":"2024-02-12T05:13:14.888902Z","shell.execute_reply":"2024-02-12T05:13:38.774045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SUBMISSION","metadata":{}},{"cell_type":"code","source":"preds_proba_sumbission = model.predict_proba(df_merged_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:13:51.357091Z","iopub.execute_input":"2024-02-12T05:13:51.357388Z","iopub.status.idle":"2024-02-12T05:13:51.365934Z","shell.execute_reply.started":"2024-02-12T05:13:51.357366Z","shell.execute_reply":"2024-02-12T05:13:51.364551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_proba_sumbission_df = pd.DataFrame(list(zip(list(df_merged_test.index),preds_proba_sumbission)),\n              columns=['case_id','score'])\npreds_proba_sumbission_df = preds_proba_sumbission_df.set_index('case_id')","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:13:51.716877Z","iopub.execute_input":"2024-02-12T05:13:51.717291Z","iopub.status.idle":"2024-02-12T05:13:51.723746Z","shell.execute_reply.started":"2024-02-12T05:13:51.717261Z","shell.execute_reply":"2024-02-12T05:13:51.722615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_proba_sumbission_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T05:13:51.937646Z","iopub.execute_input":"2024-02-12T05:13:51.937959Z","iopub.status.idle":"2024-02-12T05:13:51.945106Z","shell.execute_reply.started":"2024-02-12T05:13:51.937935Z","shell.execute_reply":"2024-02-12T05:13:51.944132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}