{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\n# !pip install \nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nimport pickle as pkl\nimport  gc\nimport glob\nfrom tqdm import tqdm \nimport os","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:01:24.095323Z","iopub.execute_input":"2024-02-12T13:01:24.096557Z","iopub.status.idle":"2024-02-12T13:01:24.103206Z","shell.execute_reply.started":"2024-02-12T13:01:24.096514Z","shell.execute_reply":"2024-02-12T13:01:24.101995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files_path = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/\"\ntest_files_path = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/\"","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:01:24.215718Z","iopub.execute_input":"2024-02-12T13:01:24.217079Z","iopub.status.idle":"2024-02-12T13:01:24.222869Z","shell.execute_reply.started":"2024-02-12T13:01:24.217029Z","shell.execute_reply":"2024-02-12T13:01:24.221545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Helper functions\ndef reduce_mem_usage(df, int_cast=True, obj_to_category=False, subset=None):\n    \"\"\"\n    Iterate through all the columns of a dataframe and modify the data type to reduce memory usage.\n    :param df: dataframe to reduce (pd.DataFrame)\n    :param int_cast: indicate if columns should be tried to be casted to int (bool)\n    :param obj_to_category: convert non-datetime related objects to category dtype (bool)\n    :param subset: subset of columns to analyse (list)\n    :return: dataset with the column dtypes adjusted (pd.DataFrame)\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024 ** 2;\n    gc.collect()\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n#     cols_none = subset if subset is  None else df.columns.tolist()\n#     for col_non in tqdm(cols_none):\n#         df[col_non] = df[col_non].fillna(-888)\n    \n    cols = subset if subset is not None else df.columns.tolist()\n\n    for col in tqdm(cols):\n        col_type = df[col].dtype\n\n        if col_type != object and col_type.name != 'category' and 'datetime' not in col_type.name:\n            df[col] = df[col].fillna(0)\n            c_min = df[col].min()\n            c_max = df[col].max()\n\n#             # test if column can be converted to an integer\n#             treat_as_int = str(col_type)[:3] == 'int'\n#             if int_cast and not treat_as_int:\n#                 treat_as_int = check_if_integer(df[col])\n                \n            treat_as_int = True\n            if treat_as_int:\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.uint8).min and c_max < np.iinfo(np.uint8).max:\n                    df[col] = df[col].astype(np.uint8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.uint16).min and c_max < np.iinfo(np.uint16).max:\n                    df[col] = df[col].astype(np.uint16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.uint32).min and c_max < np.iinfo(np.uint32).max:\n                    df[col] = df[col].astype(np.uint32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n                elif c_min > np.iinfo(np.uint64).min and c_max < np.iinfo(np.uint64).max:\n                    df[col] = df[col].astype(np.uint64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        elif 'datetime' not in col_type.name and obj_to_category:\n            df[col] = df[col].fillna('Mis')\n            df[col] = df[col].astype('category')\n    gc.collect()\n    end_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage after optimization is: {:.3f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n\ndef date_column_depth_0(df):\n    date_columns = ['date_decision'] + [x for x in df.columns if x[-1] == 'D'] \n    df[date_columns] = df[date_columns].apply(pd.to_datetime, errors='coerce')\n    df_diff = df[date_columns].apply(lambda col: (df['date_decision'] - col).dt.days)\n    df_diff.columns = [f'Diff_{col}' for col in df_diff.columns]\n    df = pd.concat([df, df_diff], axis=1)\n    return df\n\ndef gini(x):\n    total = 0\n    for i, xi in enumerate(x[:-1], 1):\n        total += np.sum(np.abs(xi - x[i:]))\n    return total / (len(x)**2 * np.mean(x))\n\ndef multi_merge(base_data,train_vs_test,data_type):\n    if train_vs_test ==  'train':\n        file_path = train_files_path\n        list_parq =  [file_path + '/' + i for i in os.listdir(file_path) if data_type in i ] \n        \n    elif train_vs_test ==  'test':\n        file_path = test_files_path\n        list_parq =  [file_path + '/' + i for i in os.listdir(file_path) if data_type in i ] \n        \n    df_i_merged = pd.DataFrame()\n    \n    for i in list_parq:\n        print(i)\n        df_i = pd.read_parquet(i)\n        df_i = reduce_mem_usage(df_i)\n        if 'num_group1' in df_i.columns: \n            df_i = df_i[df_i['num_group1'] == 0 ]\n            df_i = df_i.drop(columns = 'num_group1')\n    #         df_i_merged = df_i_merged.merge(df_i,how = 'left',on = 'case_id')\n        df_i_merged = pd.concat([df_i_merged,df_i])\n        del df_i\n        gc.collect()\n    return df_i_merged        \n","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:01:24.462504Z","iopub.execute_input":"2024-02-12T13:01:24.463020Z","iopub.status.idle":"2024-02-12T13:01:24.495849Z","shell.execute_reply.started":"2024-02-12T13:01:24.462986Z","shell.execute_reply":"2024-02-12T13:01:24.494937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train\ntrain_base_df = pd.read_parquet(train_files_path + 'train_base.parquet')\ntrain_base_df = reduce_mem_usage(train_base_df)\ntrain_base_df","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:01:24.774838Z","iopub.execute_input":"2024-02-12T13:01:24.777722Z","iopub.status.idle":"2024-02-12T13:01:25.536052Z","shell.execute_reply.started":"2024-02-12T13:01:24.777678Z","shell.execute_reply":"2024-02-12T13:01:25.534947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged = train_base_df[['case_id']]\nvariable_type_list = ['train_static_0',\n                      'train_static_cb_0',\n                      'train_applprev_1',\n                      'train_credit_bureau_a_1',\n                     'train_credit_bureau_b_1',\n                     'train_debitcard_1',\n                     'train_deposit_1',\n                     'train_person_1',\n                     'train_tax_registry_a_1',\n                     'train_tax_registry_b_1',\n                     'train_tax_registry_c_1']\nfor k in variable_type_list:\n    df_k = multi_merge(train_base_df,'train',k)\n    df_merged = df_merged.merge(df_k,how = 'outer',on = 'case_id')\n    del df_k\n    gc.collect()\n    \n    \n#Merge with Base\ndf_merged_train = train_base_df.merge(df_merged,how = 'left',on = 'case_id')\ndel df_merged\n\n#Convert date columns to difference\ndate_columns_train = [x for x in df_merged_train.columns if x[-1] == 'D']\ndf_merged_train = date_column_depth_0(df_merged_train)\n\n\ndf_merged_train = df_merged_train.drop(columns = date_columns_train)\ngc.collect()\n\n\n#Fill Missialue\n# num_cols = df_merged_train.select_dtypes(include=np.number).columns\n# df_merged_train[num_cols] = df_merged_train[num_cols].fillna(0)\n\n# object_cols = df_merged_train.select_dtypes(include='object').columns\n# df_merged_train[object_cols] = df_merged_train[object_cols].fillna('Mis')\ndf_merged_train = df_merged_train.drop_duplicates(subset= 'case_id')   \n    \n#Reindexing\nidentifier_cols = ['date_decision','MONTH']\ntarget = 'target'\n# Reindex\ndf_merged_train = df_merged_train.set_index(['case_id','WEEK_NUM']) ","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:01:27.095349Z","iopub.execute_input":"2024-02-12T13:01:27.095801Z","iopub.status.idle":"2024-02-12T13:01:45.646119Z","shell.execute_reply.started":"2024-02-12T13:01:27.095769Z","shell.execute_reply":"2024-02-12T13:01:45.643831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Remove covid data\ncovid_weeks = list(np.arange(54,64))\ndf_merged_train = df_merged_train[~df_merged_train.index.isin(covid_weeks,level = 1)]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define X,y\nX = df_merged_train.drop(columns = identifier_cols + [target])\nX = X.select_dtypes(exclude=['object'])\ny = df_merged_train['target']\n#Delete data\ndel df_merged_train\ngc.collect()\n#Pick some weeks from starting and some weeks from end as OOT\nX_oot = X[X.index.isin([0,  1,  2,  3, \n                        48, 49, 50, 51, 52,\n                        87, 88, 89,90, 91],level = 1)]\ny_oot = y[y.index.isin([0,  1,  2,  3, 48, 49,\n                        50, 51, 52,87, \n                        88, 89,90, 91],level = 1)]\n\n\nX = X[~X.index.isin([0,  1,  2,  3,\n                     48, 49, 50, 51, 52,\n                     87, 88, 89,90, 91],level = 1)]\ny = y[~y.index.isin([0,  1,  2,  3, \n                     48, 49, 50, 51, 52,\n                     87, 88, 89,90, 91],level = 1)]\n\n\n#Train test split(stratified with WEEK_NUM in index 1)\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.25, stratify= list(X.index.get_level_values(1)) , random_state=42)\nX_val, X_test, y_val, y_test = train_test_split(X_val, y_val,stratify= list(X_val.index.get_level_values(1)) ,test_size=0.50, random_state=42)\n#delete\ndel X,y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:01:45.647219Z","iopub.status.idle":"2024-02-12T13:01:45.647699Z","shell.execute_reply.started":"2024-02-12T13:01:45.647494Z","shell.execute_reply":"2024-02-12T13:01:45.647513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params= {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"zero_as_missing\":True,\n    \"max_depth\": 3,\n    \"learning_rate\": 0.1,\n    \"n_estimators\": 1000,\n    \"colsample_bytree\": 0.4, \n    \"colsample_bynode\": 0.4,\n#     \"verbose\": 1,\n    \"random_state\": 42,\n#     \"device\": \"gpu\",\n    \"early_stopping_round\": 10\n}\n\nmodel = lgb.LGBMClassifier(**params)\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_val, y_val)])","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:33:23.184376Z","iopub.execute_input":"2024-02-12T12:33:23.184920Z","iopub.status.idle":"2024-02-12T12:40:17.452662Z","shell.execute_reply.started":"2024-02-12T12:33:23.184853Z","shell.execute_reply":"2024-02-12T12:40:17.451597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_pred = model.predict_proba(X_train)[:,1] \ny_val_pred = model.predict_proba(X_val)[:,1]\ny_test_pred = model.predict_proba(X_test)[:,1]\ny_oot_pred = model.predict_proba(X_oot)[:,1]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:40:17.454674Z","iopub.execute_input":"2024-02-12T12:40:17.455024Z","iopub.status.idle":"2024-02-12T12:41:26.592028Z","shell.execute_reply.started":"2024-02-12T12:40:17.454995Z","shell.execute_reply":"2024-02-12T12:41:26.590957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\ndef predict_df(X,y):\n    preds = model.predict_proba(X)[:,1] \n    pred_df = pd.DataFrame(preds,columns = ['predict_proba'],index = X.index)\n    \n# #     pred_df.merge(y,how = 'left',on)\n# #     pred_df['Deciles'] = pd.qcut(train_pred['predict_proba'],q=10,labels = False)  \n#     pred_df = pred_df.set_index(['case_id','WEEK_NUM'])\n    pred_df = pred_df.merge(y,how= 'left',left_index = True,right_index = True) \n    pred_df = pred_df.reset_index(level = 1)\n    return pred_df\n\n\n\n\ndef gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:41:26.593692Z","iopub.execute_input":"2024-02-12T12:41:26.594556Z","iopub.status.idle":"2024-02-12T12:41:26.609338Z","shell.execute_reply.started":"2024-02-12T12:41:26.594513Z","shell.execute_reply":"2024-02-12T12:41:26.608075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train\ntrain_predict_df = predict_df(X_train,y_train)\ntrain_gini_stability = gini_stability(train_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)\n\n\n#Val\nval_predict_df = predict_df(X_val,y_val)\nval_gini_stability = gini_stability(val_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)\n\n#Test\ntest_predict_df = predict_df(X_test,y_test)\ntest_gini_stability = gini_stability(test_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)\n\n#Oot\noot_predict_df = predict_df(X_oot,y_oot)\noot_gini_stability = gini_stability(oot_predict_df, score_col=\"predict_proba\", w_fallingrate=88.0, w_resstd=-0.5)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:41:26.614233Z","iopub.execute_input":"2024-02-12T12:41:26.614589Z","iopub.status.idle":"2024-02-12T12:42:35.911246Z","shell.execute_reply.started":"2024-02-12T12:41:26.614561Z","shell.execute_reply":"2024-02-12T12:42:35.909967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_gini_stability)\nprint(val_gini_stability)\nprint(test_gini_stability)\nprint(oot_gini_stability)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:42:35.912743Z","iopub.execute_input":"2024-02-12T12:42:35.913113Z","iopub.status.idle":"2024-02-12T12:42:35.924097Z","shell.execute_reply.started":"2024-02-12T12:42:35.913081Z","shell.execute_reply":"2024-02-12T12:42:35.922664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_train = roc_auc_score(y_train,y_train_pred)\nroc_auc_val = roc_auc_score(y_val,y_val_pred)\nroc_auc_test = roc_auc_score(y_test,y_test_pred)\nroc_auc_oot = roc_auc_score(y_oot,y_oot_pred)\n\n# Ginni\nginni_train = roc_auc_train * 2 - 1\nginni_val = roc_auc_val * 2 - 1\nginni_test = roc_auc_test * 2 - 1\nginni_oot = roc_auc_oot * 2 -1","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:42:41.065260Z","iopub.execute_input":"2024-02-12T12:42:41.065729Z","iopub.status.idle":"2024-02-12T12:42:41.692272Z","shell.execute_reply.started":"2024-02-12T12:42:41.065693Z","shell.execute_reply":"2024-02-12T12:42:41.691020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(roc_auc_train)\nprint(roc_auc_val)\nprint(roc_auc_test)\nprint(roc_auc_oot)\n\nprint(ginni_train)\nprint(ginni_val)\nprint(ginni_test)\nprint(ginni_oot)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:42:43.416200Z","iopub.execute_input":"2024-02-12T12:42:43.416664Z","iopub.status.idle":"2024-02-12T12:42:43.423792Z","shell.execute_reply.started":"2024-02-12T12:42:43.416630Z","shell.execute_reply":"2024-02-12T12:42:43.422461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train,X_val,X_test,y_train,y_val,y_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T12:42:46.034653Z","iopub.execute_input":"2024-02-12T12:42:46.035132Z","iopub.status.idle":"2024-02-12T12:42:46.269995Z","shell.execute_reply.started":"2024-02-12T12:42:46.035093Z","shell.execute_reply":"2024-02-12T12:42:46.268736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test\ntest_base_df = pd.read_parquet(test_files_path + 'test_base.parquet')\ntest_base_df = reduce_mem_usage(test_base_df)\n\ndf_merged = test_base_df[['case_id']]\nvariable_type_list = ['test_static_0',\n                      'test_static_cb_0',\n                      'test_applprev_1',\n                      'test_credit_bureau_a_1',\n                     'test_credit_bureau_b_1',\n                     'test_debitcard_1',\n                     'test_deposit_1',\n                     'test_person_1',\n                     'test_tax_registry_a_1',\n                     'test_tax_registry_b_1',\n                     'test_tax_registry_c_1']\n\nfor k in variable_type_list:\n    df_k = multi_merge(test_base_df,'test',k)\n    df_merged = df_merged.merge(df_k,how = 'outer',on = 'case_id')\n    del df_k\n    gc.collect()\n    \n    \n#Merge with Base\ndf_merged_test = test_base_df.merge(df_merged,how = 'left',on = 'case_id')\ndel df_merged\n#Convert date columns to difference\ndate_columns_test = [x for x in df_merged_test.columns if x[-1] == 'D']\ndf_merged_test = date_column_depth_0(df_merged_test)\n\n\ndf_merged_test = df_merged_test.drop(columns = date_columns_test)\ngc.collect()\n\n\n#Fill Missialue\nnum_cols = df_merged_test.select_dtypes(include=np.number).columns\ndf_merged_test[num_cols] = df_merged_test[num_cols].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:02:04.515267Z","iopub.execute_input":"2024-02-12T13:02:04.515738Z","iopub.status.idle":"2024-02-12T13:02:20.224157Z","shell.execute_reply.started":"2024-02-12T13:02:04.515701Z","shell.execute_reply":"2024-02-12T13:02:20.223111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged_test['pmtamount_36A'] = df_merged_test['pmtamount_36A'].fillna(0)\ndf_merged_test['pmtamount_36A']","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:02:20.226100Z","iopub.execute_input":"2024-02-12T13:02:20.226616Z","iopub.status.idle":"2024-02-12T13:02:20.236152Z","shell.execute_reply.started":"2024-02-12T13:02:20.226584Z","shell.execute_reply":"2024-02-12T13:02:20.235047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_columns = model.feature_name_","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:02:20.237656Z","iopub.execute_input":"2024-02-12T13:02:20.238064Z","iopub.status.idle":"2024-02-12T13:02:20.244287Z","shell.execute_reply.started":"2024-02-12T13:02:20.238035Z","shell.execute_reply":"2024-02-12T13:02:20.243312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_cols = df_merged_test.select_dtypes(include='object').columns\ndf_merged_test[object_cols] = df_merged_test[object_cols].fillna('Mis')\ndf_merged_test = df_merged_test.drop_duplicates(subset= 'case_id')   \n    \n    \n#Reindexing\nidentifier_cols = ['date_decision','MONTH']\n\n#Reindex\ndf_merged_test = df_merged_test.set_index('case_id') \n\n#Define X,y\ndf_merged_test = df_merged_test.drop(columns = identifier_cols)\ndf_merged_test = df_merged_test[model_columns]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:02:20.246523Z","iopub.execute_input":"2024-02-12T13:02:20.246816Z","iopub.status.idle":"2024-02-12T13:02:20.298451Z","shell.execute_reply.started":"2024-02-12T13:02:20.246788Z","shell.execute_reply":"2024-02-12T13:02:20.297469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_proba_sumbission = model.predict_proba(df_merged_test)[:,1]\npreds_proba_sumbission_df = pd.DataFrame(list(zip(list(df_merged_test.index),preds_proba_sumbission)),\n              columns=['case_id','score'])\npreds_proba_sumbission_df = preds_proba_sumbission_df.set_index('case_id')\npreds_proba_sumbission_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T13:02:25.776381Z","iopub.execute_input":"2024-02-12T13:02:25.776821Z","iopub.status.idle":"2024-02-12T13:02:25.792103Z","shell.execute_reply.started":"2024-02-12T13:02:25.776789Z","shell.execute_reply":"2024-02-12T13:02:25.790957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}