{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8114651,"sourceType":"datasetVersion","datasetId":4746679,"isSourceIdPinned":true},{"sourceId":8383241,"sourceType":"datasetVersion","datasetId":4985614,"isSourceIdPinned":true},{"sourceId":8389604,"sourceType":"datasetVersion","datasetId":4990091,"isSourceIdPinned":true},{"sourceId":8405978,"sourceType":"datasetVersion","datasetId":4976625,"isSourceIdPinned":true},{"sourceId":8413383,"sourceType":"datasetVersion","datasetId":4949306,"isSourceIdPinned":true},{"sourceId":8494109,"sourceType":"datasetVersion","datasetId":4981626,"isSourceIdPinned":true},{"sourceId":162470947,"sourceType":"kernelVersion"},{"sourceId":178158327,"sourceType":"kernelVersion"}],"dockerImageVersionId":30665,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n!python /kaggle/usr/lib/script1/script1.py","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:05:01.867477Z","iopub.execute_input":"2024-05-24T03:05:01.868152Z","iopub.status.idle":"2024-05-24T03:07:44.182483Z","shell.execute_reply.started":"2024-05-24T03:05:01.868118Z","shell.execute_reply":"2024-05-24T03:07:44.181379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies pandas==2.0.3","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:07:44.184627Z","iopub.execute_input":"2024-05-24T03:07:44.184998Z","iopub.status.idle":"2024-05-24T03:08:14.168685Z","shell.execute_reply.started":"2024-05-24T03:07:44.184962Z","shell.execute_reply":"2024-05-24T03:08:14.167520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import StratifiedGroupKFold, StratifiedKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\n\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import KNNImputer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-24T03:08:14.170296Z","iopub.execute_input":"2024-05-24T03:08:14.170620Z","iopub.status.idle":"2024-05-24T03:08:17.310328Z","shell.execute_reply.started":"2024-05-24T03:08:14.170591Z","shell.execute_reply":"2024-05-24T03:08:17.309336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n    \n    def transform_cols(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"\n        Transforms columns in the DataFrame according to predefined rules.\n\n        Args:\n        - df (pl.DataFrame): Input DataFrame.\n\n        Returns:\n        - pl.DataFrame: DataFrame with transformed columns.\n        \"\"\"\n        if \"riskassesment_302T\" in df.columns:\n            if df[\"riskassesment_302T\"].dtype == pl.Null:\n                df = df.with_columns(\n                    [\n                        pl.Series(\n                            \"riskassesment_302T_rng\", df[\"riskassesment_302T\"], pl.UInt8\n                        ),\n                        pl.Series(\n                            \"riskassesment_302T_mean\", df[\"riskassesment_302T\"], pl.UInt8\n                        ),\n                    ]\n                )\n            else:\n                pct_low: pl.Series = (\n                    df[\"riskassesment_302T\"]\n                    .str.split(\" - \")\n                    .apply(lambda x: x[0].replace(\"%\", \"\"))\n                    .cast(pl.UInt8)\n                )\n                pct_high: pl.Series = (\n                    df[\"riskassesment_302T\"]\n                    .str.split(\" - \")\n                    .apply(lambda x: x[1].replace(\"%\", \"\"))\n                    .cast(pl.UInt8)\n                )\n\n                diff: pl.Series = pct_high - pct_low\n                avg: pl.Series = ((pct_low + pct_high) / 2).cast(pl.Float32)\n\n                del pct_high, pct_low\n                gc.collect()\n\n                df = df.with_columns(\n                    [\n                        diff.alias(\"riskassesment_302T_rng\"),\n                        avg.alias(\"riskassesment_302T_mean\"),\n                    ]\n                )\n\n            df.drop(\"riskassesment_302T\")\n\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.7:\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n                    \n        for col in df.columns:\n            if col[-1] == \"C\":\n                if (df[col] == 0).mean() > 0.9:\n                    df = df.drop(col)\n        \n        return df\n\ncategories = {}\nclass Aggregator:\n    #Please add or subtract features yourself, be aware that too many features will take up too much space.\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        \n        #--------------------------------\n        expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n\n#         expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n#         expr_median = [pl.median(col).alias(f\"median_{col}\") for col in cols]\n        #--------------------------------\n        \n        \n        return expr_max +expr_last+expr_mean+expr_var+expr_count#+expr_min+expr_median\n    \n    def cat_expr(df, test=False):\n        agg_cols = []\n        if test:\n            cat_cols = categories.keys()\n            for col in df.columns:\n                if col in cat_cols:\n                    for value in categories[col]:\n                        agg_cols += [pl.col(col).filter(pl.col(col) == value).count().alias(f\"{col}_{value}_C\")]\n        else:\n            for col in df.select([pl.col(pl.String), pl.col(pl.Boolean)]).columns:\n                values = df[col].unique().to_list()\n                if len(values) <= 10 and df[col].is_null().mean() < 0.9:\n                    try:\n                        categories[col] = list(set(categories[col] + values))\n                    except:\n                        categories[col] = values\n                    for value in values:\n                        agg_cols += [pl.col(col).filter(pl.col(col) == value).count().alias(f\"{col}_{value}_C\")]\n\n        return agg_cols\n    \n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        \n        #--------------------------------\n#         expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n#         expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        #--------------------------------\n        \n        \n        return  expr_max +expr_last+expr_mean\n    \n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        #expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n\n        return  expr_max +expr_last#+expr_count\n    \n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max +expr_last\n    \n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] \n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        \n        #--------------------------------\n        expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        #--------------------------------\n        \n        return  expr_max +expr_last+expr_count\n    \n    def get_exprs(df, test=False, cat_cols=False):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        if cat_cols:\n            exprs += Aggregator.cat_expr(df, test)\n\n        return exprs\n\ndef read_file(path, depth=None, test=False, cat_cols=False):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1,2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df, test, cat_cols)) \n    return df\n\ndef read_files(regex_path, depth=None, test=False, cat_cols=False):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df, test, cat_cols))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"diagonal_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols\n\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:17.313156Z","iopub.execute_input":"2024-05-24T03:08:17.313869Z","iopub.status.idle":"2024-05-24T03:08:17.365763Z","shell.execute_reply.started":"2024-05-24T03:08:17.313814Z","shell.execute_reply":"2024-05-24T03:08:17.364891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X[estimator.feature_name_])[:,1]*1.05 for estimator in self.estimators[15:]]\n        y_preds += [model_4.predict(X).data.squeeze()*1.75]\n        y_preds += [estimator.predict_proba(X[estimator.feature_name_])[:,1] for estimator in self.estimators[5:10]]\n        X[cat_cols] = X[cat_cols].astype(str)\n        y_preds += [estimator.predict_proba(X[estimator.feature_names_])[:,1]*1.5 for estimator in self.estimators[:5]]\n        X[cat_cols] = X[cat_cols].astype(\"category\")\n        y_preds += [estimator.predict_proba(X[estimator.feature_names_in_])[:,1] for estimator in self.estimators[10:15]]\n        return y_preds","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:17.367043Z","iopub.execute_input":"2024-05-24T03:08:17.367987Z","iopub.status.idle":"2024-05-24T03:08:17.386140Z","shell.execute_reply.started":"2024-05-24T03:08:17.367961Z","shell.execute_reply":"2024-05-24T03:08:17.385305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:17.387260Z","iopub.execute_input":"2024-05-24T03:08:17.387568Z","iopub.status.idle":"2024-05-24T03:08:17.402558Z","shell.execute_reply.started":"2024-05-24T03:08:17.387544Z","shell.execute_reply":"2024-05-24T03:08:17.401738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:17.403657Z","iopub.execute_input":"2024-05-24T03:08:17.403998Z","iopub.status.idle":"2024-05-24T03:08:17.413765Z","shell.execute_reply.started":"2024-05-24T03:08:17.403968Z","shell.execute_reply":"2024-05-24T03:08:17.413052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:17.414812Z","iopub.execute_input":"2024-05-24T03:08:17.415114Z","iopub.status.idle":"2024-05-24T03:08:18.632059Z","shell.execute_reply.started":"2024-05-24T03:08:17.415092Z","shell.execute_reply":"2024-05-24T03:08:18.631245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[\"lgbm_score\"] = joblib.load(\"/kaggle/input/home-credit-lgbm-dataset/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:18.633357Z","iopub.execute_input":"2024-05-24T03:08:18.634071Z","iopub.status.idle":"2024-05-24T03:08:18.779132Z","shell.execute_reply.started":"2024-05-24T03:08:18.634036Z","shell.execute_reply":"2024-05-24T03:08:18.778347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[\"xgb_score\"] = joblib.load(\"/kaggle/input/home-credit-xgb-dataset/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:18.783473Z","iopub.execute_input":"2024-05-24T03:08:18.783737Z","iopub.status.idle":"2024-05-24T03:08:18.946165Z","shell.execute_reply.started":"2024-05-24T03:08:18.783715Z","shell.execute_reply":"2024-05-24T03:08:18.945291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[\"cb_score\"] = joblib.load(\"/kaggle/input/home-credit-catboost-dataset/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:18.947274Z","iopub.execute_input":"2024-05-24T03:08:18.947525Z","iopub.status.idle":"2024-05-24T03:08:19.089524Z","shell.execute_reply.started":"2024-05-24T03:08:18.947503Z","shell.execute_reply":"2024-05-24T03:08:19.088492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[\"lightautoml_score\"] = joblib.load(\"/kaggle/input/home-credit-lightautoml-dataset/denselight_oof_preds.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:19.090894Z","iopub.execute_input":"2024-05-24T03:08:19.091265Z","iopub.status.idle":"2024-05-24T03:08:19.162527Z","shell.execute_reply.started":"2024-05-24T03:08:19.091232Z","shell.execute_reply":"2024-05-24T03:08:19.161596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_stability(oof_df, \"lgbm_score\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:19.163703Z","iopub.execute_input":"2024-05-24T03:08:19.163996Z","iopub.status.idle":"2024-05-24T03:08:19.957186Z","shell.execute_reply.started":"2024-05-24T03:08:19.163972Z","shell.execute_reply":"2024-05-24T03:08:19.956168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_stability(oof_df, \"xgb_score\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:19.958357Z","iopub.execute_input":"2024-05-24T03:08:19.958621Z","iopub.status.idle":"2024-05-24T03:08:20.723880Z","shell.execute_reply.started":"2024-05-24T03:08:19.958599Z","shell.execute_reply":"2024-05-24T03:08:20.722995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_stability(oof_df, \"cb_score\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:20.725145Z","iopub.execute_input":"2024-05-24T03:08:20.725491Z","iopub.status.idle":"2024-05-24T03:08:21.481457Z","shell.execute_reply.started":"2024-05-24T03:08:20.725460Z","shell.execute_reply":"2024-05-24T03:08:21.480556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_stability(oof_df, \"lightautoml_score\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:21.482782Z","iopub.execute_input":"2024-05-24T03:08:21.483163Z","iopub.status.idle":"2024-05-24T03:08:22.233880Z","shell.execute_reply.started":"2024-05-24T03:08:21.483130Z","shell.execute_reply":"2024-05-24T03:08:22.232976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categories = joblib.load(\"/kaggle/input/home-credit-feature-engineering-dataset/categories.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:22.235004Z","iopub.execute_input":"2024-05-24T03:08:22.235306Z","iopub.status.idle":"2024-05-24T03:08:22.241475Z","shell.execute_reply.started":"2024-05-24T03:08:22.235283Z","shell.execute_reply":"2024-05-24T03:08:22.240633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_model = joblib.load(\"/kaggle/input/home-credit-lgbm-dataset/model.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:22.242607Z","iopub.execute_input":"2024-05-24T03:08:22.242949Z","iopub.status.idle":"2024-05-24T03:08:23.344085Z","shell.execute_reply.started":"2024-05-24T03:08:22.242918Z","shell.execute_reply":"2024-05-24T03:08:23.343278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model = joblib.load(\"/kaggle/input/home-credit-xgb-dataset/model.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:23.345248Z","iopub.execute_input":"2024-05-24T03:08:23.345601Z","iopub.status.idle":"2024-05-24T03:08:24.506024Z","shell.execute_reply.started":"2024-05-24T03:08:23.345553Z","shell.execute_reply":"2024-05-24T03:08:24.505140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_model = joblib.load(\"/kaggle/input/home-credit-catboost-dataset/model.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:24.507355Z","iopub.execute_input":"2024-05-24T03:08:24.508386Z","iopub.status.idle":"2024-05-24T03:08:29.005723Z","shell.execute_reply.started":"2024-05-24T03:08:24.508351Z","shell.execute_reply":"2024-05-24T03:08:29.004693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"khiemht_cb_model = joblib.load(\"/kaggle/input/hc2024-lgbm-xgb-catboost/catboost_voting.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:29.006928Z","iopub.execute_input":"2024-05-24T03:08:29.007216Z","iopub.status.idle":"2024-05-24T03:08:33.295116Z","shell.execute_reply.started":"2024-05-24T03:08:29.007191Z","shell.execute_reply":"2024-05-24T03:08:33.294179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"khiemht_lgbm_model = joblib.load(\"/kaggle/input/hc2024-lgbm-xgb-catboost/lgbm_voting.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:33.296483Z","iopub.execute_input":"2024-05-24T03:08:33.297072Z","iopub.status.idle":"2024-05-24T03:08:33.845057Z","shell.execute_reply.started":"2024-05-24T03:08:33.297039Z","shell.execute_reply":"2024-05-24T03:08:33.844182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cenc = joblib.load(\"/kaggle/input/hc2024-lgbm-xgb-catboost/count_encoder.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:33.846161Z","iopub.execute_input":"2024-05-24T03:08:33.846449Z","iopub.status.idle":"2024-05-24T03:08:34.400900Z","shell.execute_reply.started":"2024-05-24T03:08:33.846424Z","shell.execute_reply":"2024-05-24T03:08:34.400027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1, True, True),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1, True),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1, True),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1, True),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2, True),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2, True),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2, True),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2, True)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:34.402035Z","iopub.execute_input":"2024-05-24T03:08:34.402564Z","iopub.status.idle":"2024-05-24T03:08:34.644928Z","shell.execute_reply.started":"2024-05-24T03:08:34.402538Z","shell.execute_reply":"2024-05-24T03:08:34.643904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)\ndel data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:34.646181Z","iopub.execute_input":"2024-05-24T03:08:34.646471Z","iopub.status.idle":"2024-05-24T03:08:34.874111Z","shell.execute_reply.started":"2024-05-24T03:08:34.646447Z","shell.execute_reply":"2024-05-24T03:08:34.873155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks = []\n\nfor path in glob(str(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\")):\n    print(path)\n    df = pl.read_parquet(path, columns=[\"case_id\", \"num_group1\", \"pmts_month_158T\", \"pmts_month_706T\"])\n    df = df.group_by([\"case_id\", \"num_group1\"]).agg(\n        pl.count(\"pmts_month_158T\").alias(\"count_pmts_month_158T\"),\n        pl.count(\"pmts_month_706T\").alias(\"count_pmts_month_706T\"),\n    ).group_by(\"case_id\").agg(\n        pl.when(pl.col(\"count_pmts_month_158T\") > 0).then(pl.col(\"count_pmts_month_158T\")).otherwise(None).min().alias(\"min_count_pmts_month_158T\"),\n        pl.when(pl.col(\"count_pmts_month_158T\") > 0).then(pl.col(\"count_pmts_month_158T\")).otherwise(None).max().alias(\"max_count_pmts_month_158T\"),\n        pl.when(pl.col(\"count_pmts_month_158T\") > 0).then(pl.col(\"count_pmts_month_158T\")).otherwise(None).mean().alias(\"mean_count_pmts_month_158T\"),\n        pl.when(pl.col(\"count_pmts_month_706T\") > 0).then(pl.col(\"count_pmts_month_706T\")).otherwise(None).min().alias(\"min_count_pmts_month_706T\"),\n        pl.when(pl.col(\"count_pmts_month_706T\") > 0).then(pl.col(\"count_pmts_month_706T\")).otherwise(None).max().alias(\"max_count_pmts_month_706T\"),\n        pl.when(pl.col(\"count_pmts_month_706T\") > 0).then(pl.col(\"count_pmts_month_706T\")).otherwise(None).mean().alias(\"mean_count_pmts_month_706T\"),\n    )\n    chunks.append(df)\n\ncredit_bureau_a_2_feats = pl.concat(chunks, how=\"diagonal_relaxed\")\ncredit_bureau_a_2_feats = credit_bureau_a_2_feats.unique(subset=[\"case_id\"])\n\ndel df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:34.875311Z","iopub.execute_input":"2024-05-24T03:08:34.875611Z","iopub.status.idle":"2024-05-24T03:08:35.068651Z","shell.execute_reply.started":"2024-05-24T03:08:34.875586Z","shell.execute_reply":"2024-05-24T03:08:35.067815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lightautoml_columns, _ = joblib.load(\"/kaggle/input/home-credit-lightautoml-dataset/train_cat_columns.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.069874Z","iopub.execute_input":"2024-05-24T03:08:35.070162Z","iopub.status.idle":"2024-05-24T03:08:35.081088Z","shell.execute_reply.started":"2024-05-24T03:08:35.070138Z","shell.execute_reply":"2024-05-24T03:08:35.079971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = set(list(xgb_model.estimators[0].feature_names_in_) + \n                        list(lightautoml_columns) + \n                        cb_model.estimators[0].feature_names_ +\n                        khiemht_lgbm_model.estimators[0].feature_name_)","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.089329Z","iopub.execute_input":"2024-05-24T03:08:35.089671Z","iopub.status.idle":"2024-05-24T03:08:35.097588Z","shell.execute_reply.started":"2024-05-24T03:08:35.089647Z","shell.execute_reply":"2024-05-24T03:08:35.096681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CAT_COLS = ['description_5085714M',\n 'education_1103M',\n 'education_88M',\n 'maritalst_385M',\n 'maritalst_893M',\n 'requesttype_4525192L',\n 'bankacctype_710L',\n 'cardtype_51L',\n 'credtype_322L',\n 'disbursementtype_67L',\n 'equalitydataagreement_891L',\n 'inittransactioncode_186L',\n 'isdebitcard_729L',\n 'lastapprcommoditycat_1041M',\n 'lastcancelreason_561M',\n 'lastrejectcommoditycat_161M',\n 'lastrejectcommodtypec_5251769M',\n 'lastrejectreason_759M',\n 'lastrejectreasonclient_4145040M',\n 'lastst_736L',\n 'opencred_647L',\n 'paytype1st_925L',\n 'paytype_783L',\n 'twobodfilling_608L',\n 'typesuite_864L',\n 'max_cancelreason_3545846M',\n 'max_education_1138M',\n 'max_postype_4733339M',\n 'max_rejectreason_755M',\n 'max_rejectreasonclient_4145042M',\n 'last_cancelreason_3545846M',\n 'last_education_1138M',\n 'last_postype_4733339M',\n 'last_rejectreason_755M',\n 'last_rejectreasonclient_4145042M',\n 'max_credacc_status_367L',\n 'max_credtype_587L',\n 'max_familystate_726L',\n 'max_inittransactioncode_279L',\n 'max_isbidproduct_390L',\n 'max_isdebitcard_527L',\n 'max_status_219L',\n 'last_credacc_status_367L',\n 'last_credtype_587L',\n 'last_familystate_726L',\n 'last_inittransactioncode_279L',\n 'last_isbidproduct_390L',\n 'last_status_219L',\n 'max_classificationofcontr_13M',\n 'max_classificationofcontr_400M',\n 'max_contractst_545M',\n 'max_contractst_964M',\n 'max_description_351M',\n 'max_financialinstitution_382M',\n 'max_financialinstitution_591M',\n 'max_purposeofcred_426M',\n 'max_purposeofcred_874M',\n 'max_subjectrole_182M',\n 'max_subjectrole_93M',\n 'last_classificationofcontr_13M',\n 'last_classificationofcontr_400M',\n 'last_contractst_545M',\n 'last_contractst_964M',\n 'last_description_351M',\n 'last_financialinstitution_382M',\n 'last_financialinstitution_591M',\n 'last_purposeofcred_426M',\n 'last_purposeofcred_874M',\n 'last_subjectrole_182M',\n 'last_subjectrole_93M',\n 'max_education_927M',\n 'max_empladdr_district_926M',\n 'max_empladdr_zipcode_114M',\n 'max_language1_981M',\n 'last_education_927M',\n 'last_empladdr_district_926M',\n 'last_empladdr_zipcode_114M',\n 'last_language1_981M',\n 'max_contaddr_matchlist_1032L',\n 'max_contaddr_smempladdr_334L',\n 'max_empl_employedtotal_800L',\n 'max_empl_industry_691L',\n 'max_familystate_447L',\n 'max_housetype_905L',\n 'max_incometype_1044T',\n 'max_relationshiptoclient_415T',\n 'max_relationshiptoclient_642T',\n 'max_remitter_829L',\n 'max_role_1084L',\n 'max_safeguarantyflag_411L',\n 'max_sex_738L',\n 'max_type_25L',\n 'last_contaddr_matchlist_1032L',\n 'last_contaddr_smempladdr_334L',\n 'last_incometype_1044T',\n 'last_relationshiptoclient_415T',\n 'last_relationshiptoclient_642T',\n 'last_remitter_829L',\n 'last_role_1084L',\n 'last_safeguarantyflag_411L',\n 'last_sex_738L',\n 'last_type_25L',\n 'max_collater_typofvalofguarant_298M',\n 'max_collater_typofvalofguarant_407M',\n 'max_collaterals_typeofguarante_359M',\n 'max_collaterals_typeofguarante_669M',\n 'max_subjectroles_name_541M',\n 'max_subjectroles_name_838M',\n 'last_collater_typofvalofguarant_298M',\n 'last_collater_typofvalofguarant_407M',\n 'last_collaterals_typeofguarante_359M',\n 'last_collaterals_typeofguarante_669M',\n 'last_subjectroles_name_541M',\n 'last_subjectroles_name_838M',\n 'max_cacccardblochreas_147M',\n 'last_cacccardblochreas_147M',\n 'max_conts_type_509L',\n 'max_credacc_cards_status_52L',\n 'last_conts_type_509L',\n 'max_conts_role_79M',\n 'max_empls_economicalst_849M',\n 'max_empls_employer_name_740M',\n 'last_conts_role_79M',\n 'last_empls_economicalst_849M',\n 'last_empls_employer_name_740M']","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.098708Z","iopub.execute_input":"2024-05-24T03:08:35.099012Z","iopub.status.idle":"2024-05-24T03:08:35.110085Z","shell.execute_reply.started":"2024-05-24T03:08:35.098988Z","shell.execute_reply":"2024-05-24T03:08:35.109260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.pipe(Pipeline.transform_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.111248Z","iopub.execute_input":"2024-05-24T03:08:35.113536Z","iopub.status.idle":"2024-05-24T03:08:35.127057Z","shell.execute_reply.started":"2024-05-24T03:08:35.113500Z","shell.execute_reply":"2024-05-24T03:08:35.126092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.join(credit_bureau_a_2_feats, how=\"left\", on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.128214Z","iopub.execute_input":"2024-05-24T03:08:35.128475Z","iopub.status.idle":"2024-05-24T03:08:35.136721Z","shell.execute_reply.started":"2024-05-24T03:08:35.128453Z","shell.execute_reply":"2024-05-24T03:08:35.135975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.select([col for col in selected_features if not col.endswith(\"_enc\") and col != \"target\"])\n\nprint(\"test data shape:\\t\", df_test.shape)\n\ndf_test, cat_cols = to_pandas(df_test, CAT_COLS)\ndf_test = reduce_mem_usage(df_test)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.137638Z","iopub.execute_input":"2024-05-24T03:08:35.137914Z","iopub.status.idle":"2024-05-24T03:08:35.656017Z","shell.execute_reply.started":"2024-05-24T03:08:35.137890Z","shell.execute_reply":"2024-05-24T03:08:35.655143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_enc = cenc.transform(df_test[CAT_COLS])\ndf_test_enc.columns = [f\"{fn}_enc\" for fn in df_test_enc.columns]","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.657722Z","iopub.execute_input":"2024-05-24T03:08:35.658105Z","iopub.status.idle":"2024-05-24T03:08:35.823554Z","shell.execute_reply.started":"2024-05-24T03:08:35.658070Z","shell.execute_reply":"2024-05-24T03:08:35.822448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = reduce_mem_usage(pd.concat([df_test, df_test_enc], axis=1))","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:35.824949Z","iopub.execute_input":"2024-05-24T03:08:35.825282Z","iopub.status.idle":"2024-05-24T03:08:36.123852Z","shell.execute_reply.started":"2024-05-24T03:08:35.825242Z","shell.execute_reply":"2024-05-24T03:08:36.122945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Selection","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies lightautoml==0.3.8","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:08:36.125151Z","iopub.execute_input":"2024-05-24T03:08:36.125444Z","iopub.status.idle":"2024-05-24T03:10:54.126900Z","shell.execute_reply.started":"2024-05-24T03:08:36.125419Z","shell.execute_reply":"2024-05-24T03:10:54.125870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:10:54.128247Z","iopub.execute_input":"2024-05-24T03:10:54.128565Z","iopub.status.idle":"2024-05-24T03:11:22.216275Z","shell.execute_reply.started":"2024-05-24T03:10:54.128534Z","shell.execute_reply":"2024-05-24T03:11:22.215459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_4 = joblib.load(\"/kaggle/input/home-credit-lightautoml-dataset/denselight_model.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:22.217460Z","iopub.execute_input":"2024-05-24T03:11:22.218257Z","iopub.status.idle":"2024-05-24T03:11:23.268881Z","shell.execute_reply.started":"2024-05-24T03:11:22.218223Z","shell.execute_reply":"2024-05-24T03:11:23.268070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submision","metadata":{}},{"cell_type":"code","source":"model = VotingModel(cb_model.estimators +\n                    lgbm_model.estimators +\n                    xgb_model.estimators +\n                    khiemht_lgbm_model.estimators)","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:23.270124Z","iopub.execute_input":"2024-05-24T03:11:23.270436Z","iopub.status.idle":"2024-05-24T03:11:23.274764Z","shell.execute_reply.started":"2024-05-24T03:11:23.270412Z","shell.execute_reply":"2024-05-24T03:11:23.273906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = model.predict_proba(df_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:23.275927Z","iopub.execute_input":"2024-05-24T03:11:23.276187Z","iopub.status.idle":"2024-05-24T03:11:28.901208Z","shell.execute_reply.started":"2024-05-24T03:11:23.276165Z","shell.execute_reply":"2024-05-24T03:11:28.900417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_catboost = df_test.copy()\nfor col in cat_cols:\n    df_test_catboost[col] = df_test_catboost[col].cat.add_categories('Missing').fillna('Missing')","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:28.902333Z","iopub.execute_input":"2024-05-24T03:11:28.902591Z","iopub.status.idle":"2024-05-24T03:11:28.973130Z","shell.execute_reply.started":"2024-05-24T03:11:28.902569Z","shell.execute_reply":"2024-05-24T03:11:28.972100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds += [estimator.predict_proba(df_test_catboost[estimator.feature_names_])[:,1]\n            for estimator in khiemht_cb_model.estimators]","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:28.974567Z","iopub.execute_input":"2024-05-24T03:11:28.975402Z","iopub.status.idle":"2024-05-24T03:11:29.172048Z","shell.execute_reply.started":"2024-05-24T03:11:28.975364Z","shell.execute_reply":"2024-05-24T03:11:29.171275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.set_index(\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:29.173268Z","iopub.execute_input":"2024-05-24T03:11:29.174121Z","iopub.status.idle":"2024-05-24T03:11:29.193584Z","shell.execute_reply.started":"2024-05-24T03:11:29.174087Z","shell.execute_reply":"2024-05-24T03:11:29.192890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pd.Series(np.mean(y_preds, axis=0), index=df_test.index)\n\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\ndf_subm[\"score\"] = y_pred\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:29.194633Z","iopub.execute_input":"2024-05-24T03:11:29.194918Z","iopub.status.idle":"2024-05-24T03:11:29.213433Z","shell.execute_reply.started":"2024-05-24T03:11:29.194893Z","shell.execute_reply":"2024-05-24T03:11:29.212594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, y, df_test=joblib.load('/kaggle/working/data.pkl')\n\nmodel = lgb.LGBMClassifier()\nmodel.fit(df_train,y)","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:11:29.214490Z","iopub.execute_input":"2024-05-24T03:11:29.214751Z","iopub.status.idle":"2024-05-24T03:13:15.246962Z","shell.execute_reply.started":"2024-05-24T03:11:29.214728Z","shell.execute_reply":"2024-05-24T03:13:15.246002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.set_index(\"case_id\")\n\ndf_test[\"is_test\"] = pd.Series(model.predict_proba(df_test[model.feature_name_])[:,1], index=df_test.index)\ndf_test.is_test","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:13:15.248314Z","iopub.execute_input":"2024-05-24T03:13:15.248776Z","iopub.status.idle":"2024-05-24T03:13:15.277800Z","shell.execute_reply.started":"2024-05-24T03:13:15.248740Z","shell.execute_reply":"2024-05-24T03:13:15.276891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"condition=df_test.is_test<0.978\n\ndf_subm.loc[condition, 'score'] = (df_subm.loc[condition, 'score'] - 0.0718).clip(0)\ndf_subm.to_csv(\"submission.csv\")\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-24T03:13:15.279017Z","iopub.execute_input":"2024-05-24T03:13:15.279314Z","iopub.status.idle":"2024-05-24T03:13:15.293808Z","shell.execute_reply.started":"2024-05-24T03:13:15.279288Z","shell.execute_reply":"2024-05-24T03:13:15.292933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}