{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":172424009,"sourceType":"kernelVersion"},{"sourceId":174929553,"sourceType":"kernelVersion"},{"sourceId":175034590,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-01T10:06:02.845581Z","iopub.execute_input":"2024-05-01T10:06:02.845860Z","iopub.status.idle":"2024-05-01T10:06:08.389934Z","shell.execute_reply.started":"2024-05-01T10:06:02.845833Z","shell.execute_reply":"2024-05-01T10:06:08.389152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.391488Z","iopub.execute_input":"2024-05-01T10:06:08.391766Z","iopub.status.idle":"2024-05-01T10:06:08.404050Z","shell.execute_reply.started":"2024-05-01T10:06:08.391741Z","shell.execute_reply":"2024-05-01T10:06:08.403182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df): #Extract the maximum and minimum values for features P and A, and add them as additional features.\n        no_care_col = ['case_id','WEEK_NUM','num_group1','num_group2']\n        cat_date_cols = [name for name, dtype in zip(df.columns, df.dtypes)  if dtype == pl.Utf8 or name[-1]==\"D\"]\n        cols = [col for col in df.columns if col not in cat_date_cols+no_care_col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n\n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def get_exprs(df): #Execute the above function and return the result.\n        maxexprs = Aggregator.num_expr(df)[0] \n        minexprs = Aggregator.num_expr(df)[1]\n        \n        varexprs = Aggregator.num_expr(df)[2] \n        \n        sumexprs = Aggregator.num_expr(df)[3] \n\n        return maxexprs, minexprs, varexprs, sumexprs\n    \ndef read_file(path, depth=None,drop_cols=[],choose_cols=[]): \n    df = pl.read_parquet(path)\n    \n    if len(drop_cols) > 0:\n        columns_to_drop = [col for col in drop_cols if col in df.columns]\n        df = df.drop(columns_to_drop)\n    if len(choose_cols) > 0:\n        columns_to_choose = [col for col in choose_cols if col in df.columns]\n        df = df[columns_to_choose]\n    \n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs, varexpres, sumexprs = Aggregator.get_exprs(df)\n#         cat_cols_and_date = [name for name, dtype in zip(df.columns, df.dtypes)  if dtype == pl.Utf8 or name[-1]==\"D\"]\n        cat_cols_and_date = [col for col in df.columns if col not in ['case_id','WEEK_NUM','target']]\n        \n        df_cat_agg = df.filter(pl.col('num_group1') == 0)[cat_cols_and_date+['case_id']]\n        df_num = df.group_by(\"case_id\").agg(*maxexprs, *minexprs, *varexpres, *sumexprs)\n        \n        df = df_cat_agg.join(df_num, on='case_id', how='inner')\n    return df\n\ndef read_files(regex_path, depth=None,drop_cols=[],choose_cols=[]):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    \n    if len(drop_cols) > 0:\n        columns_to_drop = [col for col in drop_cols if col in df.columns]\n        df = df.drop(columns_to_drop)\n    if len(choose_cols) > 0:\n        columns_to_choose = [col for col in choose_cols if col in df.columns]\n        df = df[columns_to_choose]\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs, varexpres, sumexprs = Aggregator.get_exprs(df)\n#         cat_cols_and_date = [name for name, dtype in zip(df.columns, df.dtypes)  if dtype == pl.Utf8 or name[-1]==\"D\"]\n        cat_cols_and_date = [col for col in df.columns if col not in ['case_id','WEEK_NUM','target']]\n\n        df_cat_agg = df.filter(pl.col('num_group1') == 0)[cat_cols_and_date+['case_id']]\n\n        df_num = df.group_by(\"case_id\").agg(*maxexprs, *minexprs, *varexpres, *sumexprs)\n        \n        df = df_cat_agg.join(df_num, on='case_id', how='inner')    \n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.405560Z","iopub.execute_input":"2024-05-01T10:06:08.405851Z","iopub.status.idle":"2024-05-01T10:06:08.433461Z","shell.execute_reply.started":"2024-05-01T10:06:08.405823Z","shell.execute_reply":"2024-05-01T10:06:08.432719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 ):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n\n    return df_data, cat_cols\n\ndef reduce_mem_usage(df):\n    \"\"\" Reduce memory usage by polars dataframe {df} with name {name} by changing its data types.\n        Original pandas version of this function: https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 \"\"\"\n    print(f\"Memory usage of dataframe is {round(df.estimated_size('mb'), 2)} MB\")\n    Numeric_Int_types = [pl.Int8,pl.Int16,pl.Int32,pl.Int64]\n    Numeric_Float_types = [pl.Float32,pl.Float64]    \n    for col in df.columns:\n        try:\n            col_type = df[col].dtype\n            if col_type == pl.Categorical:\n                continue\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if col_type in Numeric_Int_types:\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df = df.with_columns(df[col].cast(pl.Int8))\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df = df.with_columns(df[col].cast(pl.Int16))\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df = df.with_columns(df[col].cast(pl.Int32))\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df = df.with_columns(df[col].cast(pl.Int64))\n            elif col_type in Numeric_Float_types:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df = df.with_columns(df[col].cast(pl.Float32))\n                else:\n                    pass\n            # elif col_type == pl.Utf8:\n            #     df = df.with_columns(df[col].cast(pl.Categorical))\n            else:\n                pass\n        except:\n            pass\n    print(f\"Memory usage of dataframe became {round(df.estimated_size('mb'), 2)} MB\")\n    gc.collect()\n\n\ndef convert_dtype(type_string):\n    if \"category\" in type_string:\n        return pd.CategoricalDtype()\n    else:\n        try:\n            return getattr(pd, type_string)()\n        except:\n            return type_string","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.435936Z","iopub.execute_input":"2024-05-01T10:06:08.436190Z","iopub.status.idle":"2024-05-01T10:06:08.452283Z","shell.execute_reply.started":"2024-05-01T10:06:08.436167Z","shell.execute_reply":"2024-05-01T10:06:08.451326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.453504Z","iopub.execute_input":"2024-05-01T10:06:08.454067Z","iopub.status.idle":"2024-05-01T10:06:08.464451Z","shell.execute_reply.started":"2024-05-01T10:06:08.454036Z","shell.execute_reply":"2024-05-01T10:06:08.463701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json \n\nwith open('/kaggle/input/feature-choosen-export/important_features.json', 'r') as file:\n    important_features = json.load(file)\n    \nwith open('/kaggle/input/feature-choosen-export/not_important_features.json', 'r') as file:\n    not_important_features = json.load(file)\n    \nfeature_important_df = pd.read_csv('/kaggle/input/home-credit-feature-selection/feature_importance.csv').iloc[:,1:]\ntop_feature = feature_important_df[feature_important_df['Gain']>=0.02]\ntop_feature_select = list(set(top_feature['Feature']))\nprint(len(top_feature_select))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.465548Z","iopub.execute_input":"2024-05-01T10:06:08.466592Z","iopub.status.idle":"2024-05-01T10:06:08.516451Z","shell.execute_reply.started":"2024-05-01T10:06:08.466567Z","shell.execute_reply":"2024-05-01T10:06:08.515526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/home-credit-lightbm-tunning/dtype_train.json', 'r') as file:\n    train_dtype = json.load(file)\ntrain_cols = list(train_dtype.keys())","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.517821Z","iopub.execute_input":"2024-05-01T10:06:08.518159Z","iopub.status.idle":"2024-05-01T10:06:08.526970Z","shell.execute_reply.started":"2024-05-01T10:06:08.518129Z","shell.execute_reply":"2024-05-01T10:06:08.525760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/home-credit-lightbm-tunning/train_cat_cols.json', 'r') as file:\n    train_cat_col = json.load(file)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.528064Z","iopub.execute_input":"2024-05-01T10:06:08.528525Z","iopub.status.idle":"2024-05-01T10:06:08.538584Z","shell.execute_reply.started":"2024-05-01T10:06:08.528488Z","shell.execute_reply":"2024-05-01T10:06:08.537537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_data_store = {\n#     \"df_base\": read_file(TEST_DIR / \"test_base.csv\"),\n#     \"depth_0\": [\n#         read_file(TEST_DIR / \"test_static_cb_0.csv\",choose_cols=important_features['static_cb']),\n#         read_files(TEST_DIR / \"test_static_0_*.csv\",choose_cols=important_features['static']),\n#     ],\n#     \"depth_1\": [\n#         read_files(TEST_DIR / \"test_applprev_1_*.csv\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_a_1.csv\", 1,drop_cols=not_important_features['tax_registry_a_1']),\n#         read_file(TEST_DIR / \"test_tax_registry_b_1.csv\", 1,drop_cols=not_important_features['tax_registry_b_1']),\n#         read_file(TEST_DIR / \"test_tax_registry_c_1.csv\", 1,drop_cols=not_important_features['tax_registry_c_1']),\n#         read_file(TEST_DIR / \"test_credit_bureau_b_1.csv\", 1),\n#         read_file(TEST_DIR / \"test_other_1.csv\", 1),\n#         read_file(TEST_DIR / \"test_person_1.csv\", 1,choose_cols=important_features['person_1']),\n#         read_file(TEST_DIR / \"test_deposit_1.csv\", 1),\n#         read_file(TEST_DIR / \"test_debitcard_1.csv\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TEST_DIR / \"test_credit_bureau_b_2.csv\", 2),\n#     ]\n# }\n\n\n\n\n# print(\"test data shape:\\t\", df_test.shape)\n# print(f'the cat col is {train_cat_col}')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.539847Z","iopub.execute_input":"2024-05-01T10:06:08.540232Z","iopub.status.idle":"2024-05-01T10:06:08.545997Z","shell.execute_reply.started":"2024-05-01T10:06:08.540198Z","shell.execute_reply":"2024-05-01T10:06:08.544957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**MAKE PREDICTION**","metadata":{}},{"cell_type":"code","source":"import pickle\nclass Config():\n    seed=2024\n    num_folds=5\n    TARGET_NAME ='target'\n    batch_size=10000\n\ndef predict():\n    test_pred_pro = np.zeros((Config.num_folds+1, len(X_submission), 2))  # Placeholder for predictions\n    \n    for fold in range(Config.num_folds):\n        print(f\"fold:{fold}\")\n        with open(f'/kaggle/input/home-credit-lightbm-tunning/lgb_fold_{fold+1}.pkl', 'rb') as file:\n            model = pickle.load(file)\n        for idx in range(0, len(X_submission), Config.batch_size):\n            # Ensure that the output of model.predict() is reshaped to fit (batch_size, 2)\n            batch_predictions = model.predict_proba(X_submission[idx:idx+Config.batch_size])[:, 1]\n            # Check if the batch_predictions need reshaping\n            if batch_predictions.ndim == 1:\n                batch_predictions = batch_predictions.reshape(-1, 1)  # Convert to (n, 1)\n                batch_predictions = np.hstack([batch_predictions, batch_predictions])  # Duplicate to make it (n, 2)\n            # Assign the reshaped predictions to the correct slice\n            test_pred_pro[fold, idx:idx + Config.batch_size] = batch_predictions\n    \n    return test_pred_pro","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.549208Z","iopub.execute_input":"2024-05-01T10:06:08.549596Z","iopub.status.idle":"2024-05-01T10:06:08.559777Z","shell.execute_reply.started":"2024-05-01T10:06:08.549564Z","shell.execute_reply":"2024-05-01T10:06:08.558659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\",choose_cols=important_features['static_cb']),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\",choose_cols=important_features['static']),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1,drop_cols=not_important_features['tax_registry_a_1']),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1,drop_cols=not_important_features['tax_registry_b_1']),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1,drop_cols=not_important_features['tax_registry_c_1']),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1,choose_cols=important_features['person_1']),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ]\n}\ndf_test = feature_eng(**test_data_store)\n#     df_test = reduce_mem_usage(df_test)\ndf_test, _ = to_pandas(df_test,cat_cols=train_cat_col)\n\ndel test_data_store\ngc.collect()\n\ncase_id_test = df_test['case_id']\n# train_cat_col = list(set(test_cat_cols).intersection(set(top_feature_select)))\nX_submission = df_test[train_cols]\n\nfor col, dtype_str in train_dtype.items():\n    if dtype_str == 'category':\n        X_submission[col] = X_submission[col].astype('category')\n    else:\n        X_submission[col] = X_submission[col].astype(dtype_str)\ntest_pred_pro = predict()\ntest_preds=test_pred_pro.mean(axis=0)[:,1]\n\nsubmission = pd.DataFrame({\n    \"case_id\": case_id_test.to_numpy(),\n    \"score\": test_preds\n}).set_index('case_id')\n# submission['score'] = submission['score'].fillna(0)\nsubmission.to_csv(\"./submission.csv\")\nsubmission\n# except:\n#     print(e)\n#     submission = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv')\n#     submission = submission.set_index('case_id')\n#     submission['score'] = 0\n#     submission.to_csv(\"./submission.csv\")\n#     submission\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:08.561476Z","iopub.execute_input":"2024-05-01T10:06:08.561800Z","iopub.status.idle":"2024-05-01T10:06:09.557496Z","shell.execute_reply.started":"2024-05-01T10:06:08.561774Z","shell.execute_reply":"2024-05-01T10:06:09.556660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try:\n#     test_data_store = {\n#     \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TEST_DIR / \"test_static_cb_0.parquet\",choose_cols=important_features['static_cb']),\n#         read_files(TEST_DIR / \"test_static_0_*.parquet\",choose_cols=important_features['static']),\n#     ],\n#     \"depth_1\": [\n#         read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1,drop_cols=not_important_features['tax_registry_a_1']),\n#         read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1,drop_cols=not_important_features['tax_registry_b_1']),\n#         read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1,drop_cols=not_important_features['tax_registry_c_1']),\n#         read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_person_1.parquet\", 1,choose_cols=important_features['person_1']),\n#         read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n#     ]\n# }\n#     df_test = feature_eng(**test_data_store)\n# #     df_test = reduce_mem_usage(df_test)\n#     df_test, _ = to_pandas(df_test,cat_cols=train_cat_col)\n\n#     del test_data_store\n#     gc.collect()\n\n#     case_id_test = df_test['case_id']\n#     # train_cat_col = list(set(test_cat_cols).intersection(set(top_feature_select)))\n#     X_submission = df_test[train_cols]\n\n#     for col, dtype_str in train_dtype.items():\n#         X_submission.loc[:, col] = X_submission[col].astype(convert_dtype(dtype_str))\n#     test_pred_pro = predict()\n#     test_preds=test_pred_pro.mean(axis=0)[:,1]\n\n#     submission = pd.DataFrame({\n#         \"case_id\": case_id_test.to_numpy(),\n#         \"score\": test_preds\n#     }).set_index('case_id')\n#     submission.to_csv(\"./submission.csv\")\n    \n# except Exception as e:\n#     print(e)\n#     df_test = read_file(TEST_DIR / \"test_base.csv\")\n\n#     case_id_test = df_test['case_id']\n#     submission = pd.DataFrame({\n#         \"case_id\": case_id_test.to_numpy(),\n#         \"score\": np.zeros(len(case_id_test))\n#     }).set_index('case_id')\n#     submission.to_csv(\"./submission.csv\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:09.558933Z","iopub.execute_input":"2024-05-01T10:06:09.559205Z","iopub.status.idle":"2024-05-01T10:06:09.564664Z","shell.execute_reply.started":"2024-05-01T10:06:09.559181Z","shell.execute_reply":"2024-05-01T10:06:09.563678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_preds=test_pred_pro.mean(axis=0)[:,1]\n# submission=pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\n# submission['score']=np.clip(np.nan_to_num(test_preds,nan=0.3),0,1)\n# submission.to_csv(\"submission.csv\",index=None)\n# submission\n\n\n# print(\"trick is all you need\")\n# #如果模型的效果会随着时间的推移越变越差\n# #比如 [0.9,0.8,0.7,0.6] 那我们需要将0.9变成0.7,0.8也变成0.7,所以每个预测的结果减小的值应该不同.\n# change_range=len(test_preds)//2#对前半部分的预测效果做微调\n# change_value=0.02#效果变差多少\n# for idx in range(change_range):#idx越大,模型效果本身就越差,所以微调的程度越小.\n#     test_preds[idx]-=change_value*(1-idx/change_range)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:09.565850Z","iopub.execute_input":"2024-05-01T10:06:09.566167Z","iopub.status.idle":"2024-05-01T10:06:09.578588Z","shell.execute_reply.started":"2024-05-01T10:06:09.566136Z","shell.execute_reply":"2024-05-01T10:06:09.577867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import lightgbm as lgb\n# gbm = lgb.Booster(model_file='/kaggle/input/home-credit-lightbm-tunning/best_LGB.bin')\n# y_submission_pred = gbm.predict(X_submission, num_iteration=gbm.best_iteration)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T10:06:09.579731Z","iopub.execute_input":"2024-05-01T10:06:09.580265Z","iopub.status.idle":"2024-05-01T10:06:09.591101Z","shell.execute_reply.started":"2024-05-01T10:06:09.580233Z","shell.execute_reply":"2024-05-01T10:06:09.590274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}