{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"colab":{"provenance":[],"name":"G14 Home Credit 2024 Starter Notebook"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nimport os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\n\nimport joblib\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"id":"Mm_NGuihhazM","execution":{"iopub.status.busy":"2024-05-01T23:12:36.580429Z","iopub.execute_input":"2024-05-01T23:12:36.581203Z","iopub.status.idle":"2024-05-01T23:12:36.589678Z","shell.execute_reply.started":"2024-05-01T23:12:36.581150Z","shell.execute_reply":"2024-05-01T23:12:36.588171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n\n        df = df.drop(\"date_decision\",\"MONTH\")\n\n        return df\n\n\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df\n","metadata":{"id":"So9JN4arhazM","execution":{"iopub.status.busy":"2024-05-01T23:12:36.591887Z","iopub.execute_input":"2024-05-01T23:12:36.592338Z","iopub.status.idle":"2024-05-01T23:12:36.609252Z","shell.execute_reply.started":"2024-05-01T23:12:36.592300Z","shell.execute_reply":"2024-05-01T23:12:36.608071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:12:36.610698Z","iopub.execute_input":"2024-05-01T23:12:36.611074Z","iopub.status.idle":"2024-05-01T23:12:36.625385Z","shell.execute_reply.started":"2024-05-01T23:12:36.611023Z","shell.execute_reply":"2024-05-01T23:12:36.624252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:12:36.628054Z","iopub.execute_input":"2024-05-01T23:12:36.628517Z","iopub.status.idle":"2024-05-01T23:12:36.640289Z","shell.execute_reply.started":"2024-05-01T23:12:36.628477Z","shell.execute_reply":"2024-05-01T23:12:36.639244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:12:36.641573Z","iopub.execute_input":"2024-05-01T23:12:36.641950Z","iopub.status.idle":"2024-05-01T23:12:36.651986Z","shell.execute_reply.started":"2024-05-01T23:12:36.641921Z","shell.execute_reply":"2024-05-01T23:12:36.651022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:12:36.653408Z","iopub.execute_input":"2024-05-01T23:12:36.653803Z","iopub.status.idle":"2024-05-01T23:12:36.665226Z","shell.execute_reply.started":"2024-05-01T23:12:36.653766Z","shell.execute_reply":"2024-05-01T23:12:36.664235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"ROOT= Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR= ROOT / \"parquet_files\" / \"train\"\nTEST_DIR= ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:12:36.666645Z","iopub.execute_input":"2024-05-01T23:12:36.667252Z","iopub.status.idle":"2024-05-01T23:12:36.676326Z","shell.execute_reply.started":"2024-05-01T23:12:36.667216Z","shell.execute_reply":"2024-05-01T23:12:36.675297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Data","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:15:31.780547Z","iopub.execute_input":"2024-05-02T00:15:31.781595Z","iopub.status.idle":"2024-05-02T00:16:07.189738Z","shell.execute_reply.started":"2024-05-02T00:15:31.781544Z","shell.execute_reply":"2024-05-02T00:16:07.188577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:16:21.151929Z","iopub.execute_input":"2024-05-02T00:16:21.152354Z","iopub.status.idle":"2024-05-02T00:16:27.874571Z","shell.execute_reply.started":"2024-05-02T00:16:21.152321Z","shell.execute_reply":"2024-05-02T00:16:27.873376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:13:17.466849Z","iopub.execute_input":"2024-05-01T23:13:17.467200Z","iopub.status.idle":"2024-05-01T23:13:17.495370Z","shell.execute_reply.started":"2024-05-01T23:13:17.467158Z","shell.execute_reply":"2024-05-01T23:13:17.494399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Testing Data","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:16:31.033518Z","iopub.execute_input":"2024-05-02T00:16:31.034040Z","iopub.status.idle":"2024-05-02T00:16:31.422470Z","shell.execute_reply.started":"2024-05-02T00:16:31.033998Z","shell.execute_reply":"2024-05-02T00:16:31.421234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:16:33.320589Z","iopub.execute_input":"2024-05-02T00:16:33.320990Z","iopub.status.idle":"2024-05-02T00:16:33.370418Z","shell.execute_reply.started":"2024-05-02T00:16:33.320958Z","shell.execute_reply":"2024-05-02T00:16:33.369064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:16:37.557341Z","iopub.execute_input":"2024-05-02T00:16:37.557729Z","iopub.status.idle":"2024-05-02T00:16:39.942281Z","shell.execute_reply.started":"2024-05-02T00:16:37.557698Z","shell.execute_reply":"2024-05-02T00:16:39.941073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, category_cols = to_pandas(df_train)\ndf_test, category_cols = to_pandas(df_test, category_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:16:41.684272Z","iopub.execute_input":"2024-05-02T00:16:41.684707Z","iopub.status.idle":"2024-05-02T00:16:58.771893Z","shell.execute_reply.started":"2024-05-02T00:16:41.684675Z","shell.execute_reply":"2024-05-02T00:16:58.770466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:17:03.001756Z","iopub.execute_input":"2024-05-02T00:17:03.002583Z","iopub.status.idle":"2024-05-02T00:17:03.332684Z","shell.execute_reply.started":"2024-05-02T00:17:03.002544Z","shell.execute_reply":"2024-05-02T00:17:03.331554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ncase_ids = np.array(df_train[\"case_id\"].unique())\nnp.random.shuffle(case_ids)  # This shuffles the array in place\n\n# Split case_ids into training, test, and validation sets\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\n# Identify prediction columns\ncols_pred = [col for col in df_train.columns if col[-1].isupper() and col[:-1].islower()]\nprint(\"Prediction columns:\", cols_pred)\n\n# Define a function to extract relevant data from df_train based on case_ids\ndef from_pandas(case_ids):\n    # Ensure case_ids is a list or numpy array\n    if not isinstance(case_ids, (list, np.ndarray)):\n        raise ValueError(\"case_ids must be a list or numpy array.\")\n\n    # Filter df_train by case_ids to get required data\n    base = df_train[df_train[\"case_id\"].isin(case_ids)][[\"case_id\", \"WEEK_NUM\", \"target\"]]\n    features = df_train[df_train[\"case_id\"].isin(case_ids)][cols_pred]\n    target = df_train[df_train[\"case_id\"].isin(case_ids)][\"target\"]\n\n    return base, features, target\n\n# Extract base, X, and y for training, validation, and testing\nbase_train, X_train, y_train = from_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_pandas(case_ids_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:17:05.774139Z","iopub.execute_input":"2024-05-02T00:17:05.774569Z","iopub.status.idle":"2024-05-02T00:17:12.314589Z","shell.execute_reply.started":"2024-05-02T00:17:05.774537Z","shell.execute_reply":"2024-05-02T00:17:12.313473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:17:25.254634Z","iopub.execute_input":"2024-05-02T00:17:25.255036Z","iopub.status.idle":"2024-05-02T00:17:25.264043Z","shell.execute_reply.started":"2024-05-02T00:17:25.255002Z","shell.execute_reply":"2024-05-02T00:17:25.262917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Use for Stability score\n#X = X_train\n#y = y_train\n#weeks = base_train[\"WEEK_NUM\"]\n\n#Use for submission\nX = df_train.drop(columns=[\"target\", \"case_id\",\"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:17:20.632764Z","iopub.execute_input":"2024-05-02T00:17:20.633149Z","iopub.status.idle":"2024-05-02T00:17:21.594367Z","shell.execute_reply.started":"2024-05-02T00:17:20.633121Z","shell.execute_reply":"2024-05-02T00:17:21.593264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\n\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,\n    \"learning_rate\": 0.05,\n    \"max_bin\": 180,\n    \"n_estimators\": 1200,\n    \"colsample_bytree\": 0.8, \n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1, \n    \"reg_lambda\": 3, \n    \"extra_trees\":True,\n    \"device\": \"gpu\",\n}\n\nfitted_models = []\ncv_scores = []  \n\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    print(\"Valid week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\n    model = lgb.LGBMClassifier(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        callbacks=[lgb.log_evaluation(100), lgb.early_stopping(100)]\n    )\n\n    fitted_models.append(model)\n\n    y_pred_valid = model.predict_proba(X_valid)[:, 1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores.append(auc_score)\n\nmodel = VotingModel(fitted_models)\nprint(\"CV AUC scores: \", cv_scores)\nprint(\"Average CV AUC score: \", sum(cv_scores) / len(cv_scores))","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:17:28.145785Z","iopub.execute_input":"2024-05-02T00:17:28.146228Z","iopub.status.idle":"2024-05-02T00:27:13.181869Z","shell.execute_reply.started":"2024-05-02T00:17:28.146169Z","shell.execute_reply":"2024-05-02T00:27:13.180616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train['score'] = pd.Series(model.predict_proba(X_train)[:, 1], index=X_train.index)\n#X_train['target'] = y_train\n\n#X_valid['score'] = pd.Series(model.predict_proba(X_valid)[:, 1], index=X_valid.index)\n#X_valid['target'] = y_valid\n\n#X_test['score'] = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)\n#X_test['target'] = y_test","metadata":{"execution":{"iopub.status.busy":"2024-05-01T06:32:19.882760Z","iopub.execute_input":"2024-05-01T06:32:19.883078Z","iopub.status.idle":"2024-05-01T06:32:19.887695Z","shell.execute_reply.started":"2024-05-01T06:32:19.883052Z","shell.execute_reply":"2024-05-01T06:32:19.886678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lgb_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T06:32:19.888820Z","iopub.execute_input":"2024-05-01T06:32:19.889153Z","iopub.status.idle":"2024-05-01T06:32:19.898575Z","shell.execute_reply.started":"2024-05-01T06:32:19.889126Z","shell.execute_reply":"2024-05-01T06:32:19.897571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#base_train, X_train, y_train = from_pandas(case_ids_train)\n#base_valid, X_valid, y_valid = from_pandas(case_ids_valid)\n#base_test, X_test, y_test = from_pandas(case_ids_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:23:54.572670Z","iopub.execute_input":"2024-05-01T23:23:54.573071Z","iopub.status.idle":"2024-05-01T23:24:00.713626Z","shell.execute_reply.started":"2024-05-01T23:23:54.573040Z","shell.execute_reply":"2024-05-01T23:24:00.712600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predict and set 'score' for train, valid, and test sets\n#for base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    # Predict scores with 'predict_proba' and add them to the base DataFrame\n    #base[\"score\"] = pd.Series(model.predict_proba(X)[:, 1], index=X.index)\n\n# Calculate and print AUC scores\n#print(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}') \n#print(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}') \n#print(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}') ","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:24:04.467498Z","iopub.execute_input":"2024-05-01T23:24:04.467899Z","iopub.status.idle":"2024-05-01T23:27:45.045226Z","shell.execute_reply.started":"2024-05-01T23:24:04.467869Z","shell.execute_reply":"2024-05-01T23:27:45.044246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del model\n#del X_test\n\n#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T06:32:19.910985Z","iopub.execute_input":"2024-05-01T06:32:19.911293Z","iopub.status.idle":"2024-05-01T06:32:19.925537Z","shell.execute_reply.started":"2024-05-01T06:32:19.911270Z","shell.execute_reply":"2024-05-01T06:32:19.924543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"markdown","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n\n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x,y,1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the training set is: {stability_score_train}')\nprint(f'The stability score on the valid set is: {stability_score_valid}')\nprint(f'The stability score on the test set is: {stability_score_test}')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T06:14:00.971029Z","iopub.status.idle":"2024-05-01T06:14:00.971535Z","shell.execute_reply.started":"2024-05-01T06:14:00.971282Z","shell.execute_reply":"2024-05-01T06:14:00.971303Z"}}},{"cell_type":"code","source":"#basetest = base_test.drop(columns=['target', 'score'])\n#combined_df = pd.concat([basetest, X_test], axis=1)\n#combined_df","metadata":{"execution":{"iopub.status.busy":"2024-05-01T06:32:19.926736Z","iopub.execute_input":"2024-05-01T06:32:19.927077Z","iopub.status.idle":"2024-05-01T06:32:19.935730Z","shell.execute_reply.started":"2024-05-01T06:32:19.927042Z","shell.execute_reply":"2024-05-01T06:32:19.934783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=['WEEK_NUM'])\nX_test = X_test.set_index(\"case_id\")\n\ngb_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:34:01.432795Z","iopub.execute_input":"2024-05-02T00:34:01.433243Z","iopub.status.idle":"2024-05-02T00:34:01.558984Z","shell.execute_reply.started":"2024-05-02T00:34:01.433208Z","shell.execute_reply":"2024-05-02T00:34:01.557973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:34:06.267015Z","iopub.execute_input":"2024-05-02T00:34:06.267412Z","iopub.status.idle":"2024-05-02T00:34:06.275831Z","shell.execute_reply.started":"2024-05-02T00:34:06.267384Z","shell.execute_reply":"2024-05-02T00:34:06.274676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm[\"score\"] =gb_pred","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:34:08.040797Z","iopub.execute_input":"2024-05-02T00:34:08.041778Z","iopub.status.idle":"2024-05-02T00:34:08.047703Z","shell.execute_reply.started":"2024-05-02T00:34:08.041731Z","shell.execute_reply":"2024-05-02T00:34:08.046220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:34:10.387727Z","iopub.execute_input":"2024-05-02T00:34:10.388546Z","iopub.status.idle":"2024-05-02T00:34:10.399582Z","shell.execute_reply.started":"2024-05-02T00:34:10.388508Z","shell.execute_reply":"2024-05-02T00:34:10.398261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-01T06:32:20.114490Z","iopub.execute_input":"2024-05-01T06:32:20.115451Z","iopub.status.idle":"2024-05-01T06:32:20.123215Z","shell.execute_reply.started":"2024-05-01T06:32:20.115409Z","shell.execute_reply":"2024-05-01T06:32:20.122042Z"},"trusted":true},"execution_count":null,"outputs":[]}]}