{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7589307,"sourceType":"datasetVersion","datasetId":4417609},{"sourceId":7598984,"sourceType":"datasetVersion","datasetId":4416403}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.ensemble import AdaBoostClassifier\n\nimport joblib\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-13T11:45:08.134391Z","iopub.execute_input":"2024-03-13T11:45:08.134869Z","iopub.status.idle":"2024-03-13T11:45:08.143459Z","shell.execute_reply.started":"2024-03-13T11:45:08.134821Z","shell.execute_reply":"2024-03-13T11:45:08.142058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df):\n    for col in df.columns:\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n        if col[-1] in (\"M\"):\n            df = df.with_columns(pl.col(col).cast(pl.String).alias(col))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:45:08.994980Z","iopub.execute_input":"2024-03-13T11:45:08.995727Z","iopub.status.idle":"2024-03-13T11:45:09.005119Z","shell.execute_reply.started":"2024-03-13T11:45:08.995683Z","shell.execute_reply":"2024-03-13T11:45:09.003213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def handle_dates(df):\n    for col in df.columns:\n        if col[-1] in (\"D\"):\n            df = df.with_columns(pl.col(col).cast(pl.Date).alias(col))\n            df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n            df = df.with_columns(pl.col(col).dt.total_days())\n            \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:02.822758Z","iopub.execute_input":"2024-03-13T11:39:02.823775Z","iopub.status.idle":"2024-03-13T11:39:02.845251Z","shell.execute_reply.started":"2024-03-13T11:39:02.823724Z","shell.execute_reply":"2024-03-13T11:39:02.843949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def filter_cols(df):\n    \n    # Drop if null count of column higher than 80%\n    \n    for col in df.columns:\n        if col in [\"case_id\", \"WEEK_NUM\"]:\n            continue\n            \n        isnull = df[col].is_null().mean()\n        \n        if isnull > 0.8:\n            df = df.drop(col)\n            \n    # Drop if number of unique values of column is not between 2-100\n            \n    for col in df.columns[1:]:\n        if col in [\"case_id\", \"WEEK_NUM\"]:\n            continue\n        if df[col].dtype != pl.String:\n            continue\n            \n        freq = df[col].n_unique()\n        \n        if (freq == 1) | (freq > 100):\n            df = df.drop(col)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:02.847234Z","iopub.execute_input":"2024-03-13T11:39:02.847725Z","iopub.status.idle":"2024-03-13T11:39:02.862867Z","shell.execute_reply.started":"2024-03-13T11:39:02.847683Z","shell.execute_reply":"2024-03-13T11:39:02.861060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path):\n    df = pl.read_parquet(path)\n    df = df.pipe(set_table_dtypes)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:02.868071Z","iopub.execute_input":"2024-03-13T11:39:02.868850Z","iopub.status.idle":"2024-03-13T11:39:02.881282Z","shell.execute_reply.started":"2024-03-13T11:39:02.868717Z","shell.execute_reply":"2024-03-13T11:39:02.879263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, df_person_1, df_static, df_static_cb, df_credit_bureau_b_2):\n    df_base = (\n        df_base\n        .with_columns(\n            date_decision = pl.col(\"date_decision\").cast(pl.Date),\n            WEEK_NUM = pl.col(\"WEEK_NUM\").cast(pl.Int32),\n        )\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    \n    df_person_1 = (\n        df_person_1\n        .group_by(\"case_id\")\n        .agg(\n            [pl.max(col) for col in df_person_1.columns if col != \"case_id\"],\n        )\n    )\n    \n    df_credit_bureau_b_2 = (\n        df_credit_bureau_b_2\n        .group_by(\"case_id\")\n        .agg(\n            [pl.max(col) for col in df_credit_bureau_b_2.columns if col != \"case_id\"],\n        )\n    )\n\n    df_data = (\n        df_base\n        .join(df_person_1, how=\"left\", on=\"case_id\", suffix=\"_p1\")\n        .join(df_static, how=\"left\", on=\"case_id\", suffix=\"_s\")\n        .join(df_static_cb, how=\"left\", on=\"case_id\", suffix=\"_scb\")\n        .join(df_credit_bureau_b_2, how=\"left\", on=\"case_id\", suffix=\"cbb2\")\n    )\n    \n    return df_data","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:02.884239Z","iopub.execute_input":"2024-03-13T11:39:02.885131Z","iopub.status.idle":"2024-03-13T11:39:02.905286Z","shell.execute_reply.started":"2024-03-13T11:39:02.885068Z","shell.execute_reply":"2024-03-13T11:39:02.903782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:02.907723Z","iopub.execute_input":"2024-03-13T11:39:02.908385Z","iopub.status.idle":"2024-03-13T11:39:02.931065Z","shell.execute_reply.started":"2024-03-13T11:39:02.908338Z","shell.execute_reply":"2024-03-13T11:39:02.927726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"\nBASE_TRAIN_PATH = TRAIN_DIR / \"train_base.parquet\"\nBASE_TEST_PATH  = TRAIN_DIR / \"train_base.parquet\"\n\nLGB_PATH = \"/kaggle/input/homecredit-dataset/lgb_model.pth\"\nADA_PATH = \"/kaggle/input/homecredit-dataset/ada_model.pth\"\n\nLOAD_MODEL = True\n\nn_fold = 10","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:45:20.190074Z","iopub.execute_input":"2024-03-13T11:45:20.190748Z","iopub.status.idle":"2024-03-13T11:45:20.197751Z","shell.execute_reply.started":"2024-03-13T11:45:20.190693Z","shell.execute_reply":"2024-03-13T11:45:20.196321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_base              = read_file(TRAIN_DIR / \"train_base.parquet\")\ndf_static_cb         = read_file(TRAIN_DIR / \"train_static_cb_0.parquet\")\ndf_person_1          = read_file(TRAIN_DIR / \"train_person_1.parquet\")\ndf_credit_bureau_b_2 = read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\")\n\ndf_static = pl.concat([\n    read_file(TRAIN_DIR / \"train_static_0_0.parquet\"),\n    read_file(TRAIN_DIR / \"train_static_0_1.parquet\"),\n], how=\"vertical_relaxed\")","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:02.956062Z","iopub.execute_input":"2024-03-13T11:39:02.957220Z","iopub.status.idle":"2024-03-13T11:39:10.775884Z","shell.execute_reply.started":"2024-03-13T11:39:02.957144Z","shell.execute_reply":"2024-03-13T11:39:10.774607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(df_base, df_person_1, df_static, df_static_cb, df_credit_bureau_b_2)\ndf_train = df_train.pipe(handle_dates)\ndf_train = df_train.pipe(filter_cols)\n\ndf_train, cat_cols = to_pandas(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:10.777609Z","iopub.execute_input":"2024-03-13T11:39:10.778082Z","iopub.status.idle":"2024-03-13T11:39:40.144194Z","shell.execute_reply.started":"2024-03-13T11:39:10.778035Z","shell.execute_reply":"2024-03-13T11:39:40.142941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"base shape:\\t\\t\", df_base.shape)\nprint(\"person_1 shape:\\t\\t\", df_person_1.shape)\nprint(\"static shape:\\t\\t\", df_static.shape)\nprint(\"static_cb shape:\\t\", df_static_cb.shape)\nprint(\"credit_bureau_b_2 shape:\", df_credit_bureau_b_2.shape)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:40.146230Z","iopub.execute_input":"2024-03-13T11:39:40.146725Z","iopub.status.idle":"2024-03-13T11:39:40.155749Z","shell.execute_reply.started":"2024-03-13T11:39:40.146680Z","shell.execute_reply":"2024-03-13T11:39:40.154241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_base              = read_file(TEST_DIR / \"test_base.parquet\")\ndf_static_cb         = read_file(TEST_DIR / \"test_static_cb_0.parquet\")\ndf_person_1          = read_file(TEST_DIR / \"test_person_1.parquet\")\ndf_credit_bureau_b_2 = read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\")\n\ndf_static = pl.concat([\n    read_file(TEST_DIR / \"test_static_0_0.parquet\"),\n    read_file(TEST_DIR / \"test_static_0_1.parquet\"),\n    read_file(TEST_DIR / \"test_static_0_2.parquet\"),\n], how=\"vertical_relaxed\")","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:40.157838Z","iopub.execute_input":"2024-03-13T11:39:40.158337Z","iopub.status.idle":"2024-03-13T11:39:40.580252Z","shell.execute_reply.started":"2024-03-13T11:39:40.158290Z","shell.execute_reply":"2024-03-13T11:39:40.579046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(df_base, df_person_1, df_static, df_static_cb, df_credit_bureau_b_2)\ndf_test = df_test.pipe(handle_dates)\ndf_test = df_test.select(df_train.columns.drop(\"target\"))\n\ndf_test, _ = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:40.581682Z","iopub.execute_input":"2024-03-13T11:39:40.582006Z","iopub.status.idle":"2024-03-13T11:39:40.652578Z","shell.execute_reply.started":"2024-03-13T11:39:40.581979Z","shell.execute_reply":"2024-03-13T11:39:40.651196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"base shape:\\t\\t\", df_base.shape)\nprint(\"person_1 shape:\\t\\t\", df_person_1.shape)\nprint(\"static shape:\\t\\t\", df_static.shape)\nprint(\"static_cb shape:\\t\", df_static_cb.shape)\nprint(\"credit_bureau_b_2 shape:\", df_credit_bureau_b_2.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:39:40.656367Z","iopub.execute_input":"2024-03-13T11:39:40.656774Z","iopub.status.idle":"2024-03-13T11:39:40.664409Z","shell.execute_reply.started":"2024-03-13T11:39:40.656742Z","shell.execute_reply":"2024-03-13T11:39:40.663088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Memory Cleaning","metadata":{}},{"cell_type":"code","source":"del df_base\ndel df_person_1\ndel df_static\ndel df_static_cb\ndel df_credit_bureau_b_2\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:45:31.199433Z","iopub.execute_input":"2024-03-13T11:45:31.200014Z","iopub.status.idle":"2024-03-13T11:45:31.400089Z","shell.execute_reply.started":"2024-03-13T11:45:31.199950Z","shell.execute_reply":"2024-03-13T11:45:31.397985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### EDA","metadata":{}},{"cell_type":"code","source":"print(\"Train is duplicated:\\t\", df_train[\"case_id\"].duplicated().any())\nprint(\"Train Week Range:\\t\", (df_train[\"WEEK_NUM\"].min(), df_train[\"WEEK_NUM\"].max()))\n\nprint()\n\nprint(\"Test is duplicated:\\t\", df_test[\"case_id\"].duplicated().any())\nprint(\"Test Week Range:\\t\", (df_test[\"WEEK_NUM\"].min(), df_test[\"WEEK_NUM\"].max()))","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:45:32.589912Z","iopub.execute_input":"2024-03-13T11:45:32.590629Z","iopub.status.idle":"2024-03-13T11:45:32.645601Z","shell.execute_reply.started":"2024-03-13T11:45:32.590590Z","shell.execute_reply":"2024-03-13T11:45:32.644620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(\n    data=df_train,\n    x=\"WEEK_NUM\",\n    y=\"target\",\n)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:45:34.971027Z","iopub.execute_input":"2024-03-13T11:45:34.971935Z","iopub.status.idle":"2024-03-13T11:45:55.772542Z","shell.execute_reply.started":"2024-03-13T11:45:34.971893Z","shell.execute_reply":"2024-03-13T11:45:55.771607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training","metadata":{}},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:47:30.787687Z","iopub.execute_input":"2024-03-13T11:47:30.788131Z","iopub.status.idle":"2024-03-13T11:47:30.798815Z","shell.execute_reply.started":"2024-03-13T11:47:30.788093Z","shell.execute_reply":"2024-03-13T11:47:30.797164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn\nfrom scipy import stats\n\ndef hc_metric(true, pred, week):\n    index = week.argsort()\n    true = true[index]\n    pred = pred[index]\n\n    uniq = np.unique(week[index], return_index=True)\n    week_unique, week_index = uniq[0],uniq[1][1:]\n    grouped_true = np.split(true, week_index)\n    grouped_pred = np.split(pred, week_index)\n\n    ginis = np.zeros(len(week_unique))\n    for i, (true,pred) in enumerate(zip(grouped_true,grouped_pred)):\n        gini = sklearn.metrics.roc_auc_score(true, pred)*2-1\n        ginis[i] = gini\n\n    slope, intercept, _, _, _ = stats.linregress(week_unique,ginis)\n    residuals = ginis - (slope*week_unique + intercept)\n\n    return np.mean(ginis) + 88.0 * min(0,slope) - 0.5 * np.std(residuals)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:47:31.597685Z","iopub.execute_input":"2024-03-13T11:47:31.598181Z","iopub.status.idle":"2024-03-13T11:47:31.610102Z","shell.execute_reply.started":"2024-03-13T11:47:31.598143Z","shell.execute_reply":"2024-03-13T11:47:31.608543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not LOAD_MODEL:\n    X = df_train.drop(columns=[\"target\", \"case_id\", \"date_decision\", \"WEEK_NUM\", \"MONTH\"])\n    y = df_train[\"target\"]\n    weeks = df_train[\"WEEK_NUM\"]\n\n    cv = StratifiedGroupKFold(n_splits=n_fold, shuffle=False)\n\n    params = {\n        \"boosting_type\": \"gbdt\",\n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        \"max_depth\": 8,\n        \"learning_rate\": 0.05,\n        \"n_estimators\": 1000,\n        \"colsample_bytree\": 0.8, \n        \"colsample_bynode\": 0.8,\n        \"class_weight\": \"balanced\",\n        \"verbose\": -1,\n    }\n\n    lgb_models = []\n    ada_models = []\n\n    \n    \n    for fold, (idx_train, idx_valid) in enumerate(cv.split(X, y, groups=weeks)):\n        print(f'***** Fold{fold+1} *****')\n        X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n        X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n        print(\"Valid week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\n        lgb_model = lgb.LGBMClassifier(**params)\n        lgb_model.fit(\n            X_train, y_train,\n            eval_set=[(X_valid, y_valid)],\n            callbacks=[lgb.log_evaluation(50), lgb.early_stopping(100, verbose=-1)]\n        )\n#         ada_model = AdaBoostClassifier()\n#         ada_model.fit(X_train, y_train)\n\n        lgb_models.append(lgb_model)\n#         ada_models.append(ada_model)\n        if not LOAD_MODEL:\n            joblib.dump(lgb_model, f\"lgb_best_model{fold+1}.pth\")\n#             joblib.dump(ada_model, f\"ada_best_model{fold+1}.pth\")\n\n        \n    lgb_model = VotingModel(lgb_models)\n#     ada_model = VotingModel(ada_models)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:47:32.304388Z","iopub.execute_input":"2024-03-13T11:47:32.304861Z","iopub.status.idle":"2024-03-13T11:47:32.318801Z","shell.execute_reply.started":"2024-03-13T11:47:32.304824Z","shell.execute_reply":"2024-03-13T11:47:32.316936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Load Model & Predict","metadata":{}},{"cell_type":"code","source":"lgb_model_pathes = glob(\"/kaggle/input/homecredit-lgbmbaselinemodel-10fold/*\")\n\nX_test = df_test.drop(columns=[\"date_decision\", \"WEEK_NUM\", \"MONTH\"])\nX_test = X_test.set_index(\"case_id\")\ny_pred = np.ndarray((len(X_test), n_fold))\n\nif LOAD_MODEL:\n    for i, lgb_path in enumerate(lgb_model_pathes):\n        lgb_model = joblib.load(lgb_path)\n    #     ada_model = joblib.load(ada_path)   \n    # y_pred = pd.Series(lgb_model.predict_proba(X_test)[:, 1]*0.825 + ada_model.predict_proba(X_test)[:, 1]*0.125, index=X_test.index)\n        y_pred[:, i] = pd.Series(lgb_model.predict_proba(X_test)[:, 1], index=X_test.index)\nelse:\n    for i in range(n_fold):\n        lgb_model = joblib.load(f\"lgb_best_model{fold+1}.pth\")\n    #     ada_model = joblib.load(f\"ada_best_model{fold+1}.pth\")   \n    # y_pred = pd.Series(lgb_model.predict_proba(X_test)[:, 1]*0.825 + ada_model.predict_proba(X_test)[:, 1]*0.125, index=X_test.index)\n        y_pred[:, i] = pd.Series(lgb_model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:47:33.953872Z","iopub.execute_input":"2024-03-13T11:47:33.954284Z","iopub.status.idle":"2024-03-13T11:47:35.170325Z","shell.execute_reply.started":"2024-03-13T11:47:33.954257Z","shell.execute_reply":"2024-03-13T11:47:35.169475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:47:37.549304Z","iopub.execute_input":"2024-03-13T11:47:37.550787Z","iopub.status.idle":"2024-03-13T11:47:37.560288Z","shell.execute_reply.started":"2024-03-13T11:47:37.550739Z","shell.execute_reply":"2024-03-13T11:47:37.558752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Submission","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv\")\nsub = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\", dtype={\"case_id\": int}).set_index(\"case_id\")\n\nsub['score'] = y_pred\nSHIFT = 0.025\nweek_num = list(test[\"WEEK_NUM\"])\nsub[\"WEEK_NUM\"] = week_num\ncondition = sub[\"WEEK_NUM\"] < (sub[\"WEEK_NUM\"].max() - sub[\"WEEK_NUM\"].min())/2 + sub[\"WEEK_NUM\"].min()\nsub.loc[condition, 'score'] = (sub.loc[condition, 'score'] - SHIFT).clip(0)\ndel sub[\"WEEK_NUM\"]\nsub.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:49:33.454355Z","iopub.execute_input":"2024-03-13T11:49:33.454761Z","iopub.status.idle":"2024-03-13T11:49:33.473730Z","shell.execute_reply.started":"2024-03-13T11:49:33.454729Z","shell.execute_reply":"2024-03-13T11:49:33.472048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:49:34.458760Z","iopub.execute_input":"2024-03-13T11:49:34.459215Z","iopub.status.idle":"2024-03-13T11:49:34.476416Z","shell.execute_reply.started":"2024-03-13T11:49:34.459182Z","shell.execute_reply":"2024-03-13T11:49:34.475059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred.mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:49:47.119870Z","iopub.execute_input":"2024-03-13T11:49:47.123464Z","iopub.status.idle":"2024-03-13T11:49:47.150401Z","shell.execute_reply.started":"2024-03-13T11:49:47.123348Z","shell.execute_reply":"2024-03-13T11:49:47.149022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm","metadata":{"execution":{"iopub.status.busy":"2024-03-13T11:49:56.086859Z","iopub.execute_input":"2024-03-13T11:49:56.087308Z","iopub.status.idle":"2024-03-13T11:49:56.100454Z","shell.execute_reply.started":"2024-03-13T11:49:56.087273Z","shell.execute_reply":"2024-03-13T11:49:56.099221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}