{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":7592683,"sourceType":"datasetVersion","datasetId":4419512}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.ensemble import AdaBoostClassifier\n\nimport joblib\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-09T05:53:28.051367Z","iopub.execute_input":"2024-02-09T05:53:28.052557Z","iopub.status.idle":"2024-02-09T05:53:30.402856Z","shell.execute_reply.started":"2024-02-09T05:53:28.052515Z","shell.execute_reply":"2024-02-09T05:53:30.401964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df):\n    for col in df.columns:\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n        if col[-1] in (\"M\"):\n            df = df.with_columns(pl.col(col).cast(pl.String).alias(col))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:31.862186Z","iopub.execute_input":"2024-02-09T05:53:31.862784Z","iopub.status.idle":"2024-02-09T05:53:31.870892Z","shell.execute_reply.started":"2024-02-09T05:53:31.862742Z","shell.execute_reply":"2024-02-09T05:53:31.868665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def handle_dates(df):\n    for col in df.columns:\n        if col[-1] in (\"D\"):\n            df = df.with_columns(pl.col(col).cast(pl.Date).alias(col))\n            df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n            df = df.with_columns(pl.col(col).dt.total_days())\n            \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:32.868013Z","iopub.execute_input":"2024-02-09T05:53:32.868430Z","iopub.status.idle":"2024-02-09T05:53:32.875613Z","shell.execute_reply.started":"2024-02-09T05:53:32.868401Z","shell.execute_reply":"2024-02-09T05:53:32.874169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def filter_cols(df):\n    \n    # Drop if null count of column higher than 80%\n    \n    for col in df.columns:\n        if col in [\"case_id\", \"WEEK_NUM\"]:\n            continue\n            \n        isnull = df[col].is_null().mean()\n        \n        if isnull > 0.8:\n            df = df.drop(col)\n            \n    # Drop if number of unique values of column is not between 2-100\n            \n    for col in df.columns[1:]:\n        if col in [\"case_id\", \"WEEK_NUM\"]:\n            continue\n        if df[col].dtype != pl.String:\n            continue\n            \n        freq = df[col].n_unique()\n        \n        if (freq == 1) | (freq > 100):\n            df = df.drop(col)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:33.721421Z","iopub.execute_input":"2024-02-09T05:53:33.721837Z","iopub.status.idle":"2024-02-09T05:53:33.729080Z","shell.execute_reply.started":"2024-02-09T05:53:33.721805Z","shell.execute_reply":"2024-02-09T05:53:33.727998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path):\n    df = pl.read_parquet(path)\n    df = df.pipe(set_table_dtypes)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:34.389634Z","iopub.execute_input":"2024-02-09T05:53:34.390031Z","iopub.status.idle":"2024-02-09T05:53:34.395335Z","shell.execute_reply.started":"2024-02-09T05:53:34.390000Z","shell.execute_reply":"2024-02-09T05:53:34.394165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, df_person_1, df_static, df_static_cb, df_credit_bureau_b_2):\n    df_base = (\n        df_base\n        .with_columns(\n            date_decision = pl.col(\"date_decision\").cast(pl.Date),\n            WEEK_NUM = pl.col(\"WEEK_NUM\").cast(pl.Int32),\n        )\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    \n    df_person_1 = (\n        df_person_1\n        .group_by(\"case_id\")\n        .agg(\n            [pl.max(col) for col in df_person_1.columns if col != \"case_id\"],\n        )\n    )\n    \n    df_credit_bureau_b_2 = (\n        df_credit_bureau_b_2\n        .group_by(\"case_id\")\n        .agg(\n            [pl.max(col) for col in df_credit_bureau_b_2.columns if col != \"case_id\"],\n        )\n    )\n\n    df_data = (\n        df_base\n        .join(df_person_1, how=\"left\", on=\"case_id\", suffix=\"_p1\")\n        .join(df_static, how=\"left\", on=\"case_id\", suffix=\"_s\")\n        .join(df_static_cb, how=\"left\", on=\"case_id\", suffix=\"_scb\")\n        .join(df_credit_bureau_b_2, how=\"left\", on=\"case_id\", suffix=\"cbb2\")\n    )\n    \n    return df_data","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:35.150292Z","iopub.execute_input":"2024-02-09T05:53:35.150719Z","iopub.status.idle":"2024-02-09T05:53:35.159774Z","shell.execute_reply.started":"2024-02-09T05:53:35.150689Z","shell.execute_reply":"2024-02-09T05:53:35.158487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:35.832117Z","iopub.execute_input":"2024-02-09T05:53:35.832492Z","iopub.status.idle":"2024-02-09T05:53:35.839217Z","shell.execute_reply.started":"2024-02-09T05:53:35.832465Z","shell.execute_reply":"2024-02-09T05:53:35.837481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"\nBASE_TRAIN_PATH = TRAIN_DIR / \"train_base.parquet\"\nBASE_TEST_PATH  = TRAIN_DIR / \"train_base.parquet\"\n\nLOAD_MODEL = True\n\nn_fold = 5","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:36.904689Z","iopub.execute_input":"2024-02-09T05:53:36.905274Z","iopub.status.idle":"2024-02-09T05:53:36.909981Z","shell.execute_reply.started":"2024-02-09T05:53:36.905234Z","shell.execute_reply":"2024-02-09T05:53:36.909201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_base              = read_file(TRAIN_DIR / \"train_base.parquet\")\ndf_static_cb         = read_file(TRAIN_DIR / \"train_static_cb_0.parquet\")\ndf_person_1          = read_file(TRAIN_DIR / \"train_person_1.parquet\")\ndf_credit_bureau_b_2 = read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\")\n\ndf_static = pl.concat([\n    read_file(TRAIN_DIR / \"train_static_0_0.parquet\"),\n    read_file(TRAIN_DIR / \"train_static_0_1.parquet\"),\n], how=\"vertical_relaxed\")","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:38.193032Z","iopub.execute_input":"2024-02-09T05:53:38.193599Z","iopub.status.idle":"2024-02-09T05:53:44.050117Z","shell.execute_reply.started":"2024-02-09T05:53:38.193554Z","shell.execute_reply":"2024-02-09T05:53:44.049100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(df_base, df_person_1, df_static, df_static_cb, df_credit_bureau_b_2)\ndf_train = df_train.pipe(handle_dates)\ndf_train = df_train.pipe(filter_cols)\n\ndf_train = df_train.to_pandas()\n# df_train, cat_cols = to_pandas(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:44.051983Z","iopub.execute_input":"2024-02-09T05:53:44.052380Z","iopub.status.idle":"2024-02-09T05:53:58.976483Z","shell.execute_reply.started":"2024-02-09T05:53:44.052342Z","shell.execute_reply":"2024-02-09T05:53:58.975558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"base shape:\\t\\t\", df_base.shape)\nprint(\"person_1 shape:\\t\\t\", df_person_1.shape)\nprint(\"static shape:\\t\\t\", df_static.shape)\nprint(\"static_cb shape:\\t\", df_static_cb.shape)\nprint(\"credit_bureau_b_2 shape:\", df_credit_bureau_b_2.shape)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:58.977861Z","iopub.execute_input":"2024-02-09T05:53:58.978174Z","iopub.status.idle":"2024-02-09T05:53:58.984308Z","shell.execute_reply.started":"2024-02-09T05:53:58.978148Z","shell.execute_reply":"2024-02-09T05:53:58.983355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_base              = read_file(TEST_DIR / \"test_base.parquet\")\ndf_static_cb         = read_file(TEST_DIR / \"test_static_cb_0.parquet\")\ndf_person_1          = read_file(TEST_DIR / \"test_person_1.parquet\")\ndf_credit_bureau_b_2 = read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\")\n\ndf_static = pl.concat([\n    read_file(TEST_DIR / \"test_static_0_0.parquet\"),\n    read_file(TEST_DIR / \"test_static_0_1.parquet\"),\n    read_file(TEST_DIR / \"test_static_0_2.parquet\"),\n], how=\"vertical_relaxed\")","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:58.986590Z","iopub.execute_input":"2024-02-09T05:53:58.986997Z","iopub.status.idle":"2024-02-09T05:53:59.330386Z","shell.execute_reply.started":"2024-02-09T05:53:58.986967Z","shell.execute_reply":"2024-02-09T05:53:59.329284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(df_base, df_person_1, df_static, df_static_cb, df_credit_bureau_b_2)\ndf_test = df_test.pipe(handle_dates)\ndf_test = df_test.select(df_train.columns.drop(\"target\"))\ndf_test = df_test.to_pandas()\n# df_test, _ = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:59.331672Z","iopub.execute_input":"2024-02-09T05:53:59.332065Z","iopub.status.idle":"2024-02-09T05:53:59.366698Z","shell.execute_reply.started":"2024-02-09T05:53:59.332029Z","shell.execute_reply":"2024-02-09T05:53:59.365053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"base shape:\\t\\t\", df_base.shape)\nprint(\"person_1 shape:\\t\\t\", df_person_1.shape)\nprint(\"static shape:\\t\\t\", df_static.shape)\nprint(\"static_cb shape:\\t\", df_static_cb.shape)\nprint(\"credit_bureau_b_2 shape:\", df_credit_bureau_b_2.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:59.368231Z","iopub.execute_input":"2024-02-09T05:53:59.368647Z","iopub.status.idle":"2024-02-09T05:53:59.375504Z","shell.execute_reply.started":"2024-02-09T05:53:59.368612Z","shell.execute_reply":"2024-02-09T05:53:59.374454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Memory Cleaning","metadata":{}},{"cell_type":"code","source":"del df_base\ndel df_person_1\ndel df_static\ndel df_static_cb\ndel df_credit_bureau_b_2\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:59.376827Z","iopub.execute_input":"2024-02-09T05:53:59.377108Z","iopub.status.idle":"2024-02-09T05:53:59.507722Z","shell.execute_reply.started":"2024-02-09T05:53:59.377085Z","shell.execute_reply":"2024-02-09T05:53:59.506693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### EDA","metadata":{}},{"cell_type":"code","source":"print(\"Train is duplicated:\\t\", df_train[\"case_id\"].duplicated().any())\nprint(\"Train Week Range:\\t\", (df_train[\"WEEK_NUM\"].min(), df_train[\"WEEK_NUM\"].max()))\n\nprint()\n\nprint(\"Test is duplicated:\\t\", df_test[\"case_id\"].duplicated().any())\nprint(\"Test Week Range:\\t\", (df_test[\"WEEK_NUM\"].min(), df_test[\"WEEK_NUM\"].max()))","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:59.509462Z","iopub.execute_input":"2024-02-09T05:53:59.510007Z","iopub.status.idle":"2024-02-09T05:53:59.570507Z","shell.execute_reply.started":"2024-02-09T05:53:59.509972Z","shell.execute_reply":"2024-02-09T05:53:59.569434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(\n    data=df_train,\n    x=\"WEEK_NUM\",\n    y=\"target\",\n)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:13.766874Z","iopub.status.idle":"2024-02-09T05:53:13.767605Z","shell.execute_reply.started":"2024-02-09T05:53:13.767115Z","shell.execute_reply":"2024-02-09T05:53:13.767137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training","metadata":{}},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:53:59.572019Z","iopub.execute_input":"2024-02-09T05:53:59.574342Z","iopub.status.idle":"2024-02-09T05:53:59.580818Z","shell.execute_reply.started":"2024-02-09T05:53:59.574311Z","shell.execute_reply":"2024-02-09T05:53:59.579644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn\nfrom scipy import stats\n\ndef hc_metric(true, pred, week):\n    index = week.argsort()\n    true = true[index]\n    pred = pred[index]\n\n    uniq = np.unique(week[index], return_index=True)\n    week_unique, week_index = uniq[0],uniq[1][1:]\n    grouped_true = np.split(true, week_index)\n    grouped_pred = np.split(pred, week_index)\n\n    ginis = np.zeros(len(week_unique))\n    for i, (true,pred) in enumerate(zip(grouped_true,grouped_pred)):\n        gini = sklearn.metrics.roc_auc_score(true, pred)*2-1\n        ginis[i] = gini\n\n    slope, intercept, _, _, _ = stats.linregress(week_unique,ginis)\n    residuals = ginis - (slope*week_unique + intercept)\n\n    return np.mean(ginis) + 88.0 * min(0,slope) - 0.5 * np.std(residuals)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:54:02.736385Z","iopub.execute_input":"2024-02-09T05:54:02.736856Z","iopub.status.idle":"2024-02-09T05:54:02.747221Z","shell.execute_reply.started":"2024-02-09T05:54:02.736820Z","shell.execute_reply":"2024-02-09T05:54:02.744764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nobj_cols = df_train.select_dtypes('object').columns\n\nfor col in obj_cols:\n#     print(col)\n    try:\n        df_train[col] = df_train[col].astype(bool)\n    except:\n        le = LabelEncoder()\n        df_train[col] = le.fit_transform(df_train[col].fillna('NaN'))\n        df_test[col] = le.transform(df_test[col].fillna('NaN'))\n        del le","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:54:06.146332Z","iopub.execute_input":"2024-02-09T05:54:06.146784Z","iopub.status.idle":"2024-02-09T05:54:09.113554Z","shell.execute_reply.started":"2024-02-09T05:54:06.146742Z","shell.execute_reply":"2024-02-09T05:54:09.112636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not LOAD_MODEL:\n    X = df_train.drop(columns=[\"target\", \"case_id\", \"date_decision\", \"WEEK_NUM\", \"MONTH\"])\n    y = df_train[\"target\"]\n    weeks = df_train[\"WEEK_NUM\"]\n\n    cv = StratifiedGroupKFold(n_splits=n_fold, shuffle=False)\n\n    params = {\n        \"boosting_type\": \"gbdt\",\n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        \"max_depth\": 8,\n        \"learning_rate\": 0.05,\n        \"n_estimators\": 1000,\n        \"colsample_bytree\": 0.8, \n        \"colsample_bynode\": 0.8,\n        \"verbose\": -1,\n    }\n\n    lgb_models = []\n    ada_models = []\n\n    \n    \n    for fold, (idx_train, idx_valid) in enumerate(cv.split(X, y, groups=weeks)):\n        print(f'***** Fold{fold+1} *****')\n        X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n        X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n        print(\"Valid week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\n#         lgb_model = lgb.LGBMClassifier(**params)\n#         lgb_model.fit(\n#             X_train, y_train,\n#             eval_set=[(X_valid, y_valid)],\n#             callbacks=[lgb.early_stopping(100, verbose=-1)]\n#         )\n        ada_model = AdaBoostClassifier(n_estimators=100, learning_rate=.5, random_state=0)\n        ada_model.fit(X_train.fillna(-9999), y_train)\n\n#         lgb_models.append(lgb_model)\n        ada_models.append(ada_model)\n        if not LOAD_MODEL:\n#             joblib.dump(lgb_model, f\"lgb_best_model{fold+1}.pth\")\n            joblib.dump(ada_model, f\"ada_best_model{fold+1}.pth\")\n\n        \n#     lgb_model = VotingModel(lgb_models)\n    ada_model = VotingModel(ada_models)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:54:10.112986Z","iopub.execute_input":"2024-02-09T05:54:10.113420Z","iopub.status.idle":"2024-02-09T05:54:10.124224Z","shell.execute_reply.started":"2024-02-09T05:54:10.113387Z","shell.execute_reply":"2024-02-09T05:54:10.123248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lgb_model_pathes = glob(\"/kaggle/input/homecredit-lgbmbaseline-model/*\")\nada_model_pathes = glob(\"/kaggle/input/homecredit-adaboostbaseline-model/*\")\n\nX_test = df_test.drop(columns=[\"date_decision\", \"WEEK_NUM\", \"MONTH\"])\nX_test = X_test.set_index(\"case_id\")\ny_pred = np.ndarray((len(X_test), n_fold))\n\nif LOAD_MODEL:\n    for i, ada_path in enumerate(ada_model_pathes):\n#         lgb_model = joblib.load(lgb_path)\n        ada_model = joblib.load(ada_path)\n    # y_pred = pd.Series(lgb_model.predict_proba(X_test)[:, 1]*0.825 + ada_model.predict_proba(X_test)[:, 1]*0.125, index=X_test.index)\n#         y_pred[:, i] = pd.Series(lgb_model.predict_proba(X_test)[:, 1], index=X_test.index)\n        y_pred[:, i] = pd.Series(ada_model.predict_proba(X_test.fillna(0))[:, 1], index=X_test.index) \nelse:\n    for i in range(n_fold):\n#         lgb_model = joblib.load(f\"lgb_best_model{fold+1}.pth\")\n        ada_model = joblib.load(f\"ada_best_model{fold+1}.pth\")   \n    # y_pred = pd.Series(lgb_model.predict_proba(X_test)[:, 1]*0.825 + ada_model.predict_proba(X_test)[:, 1]*0.125, index=X_test.index)\n        y_pred[:, i] = pd.Series(lgb_model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:55:08.859583Z","iopub.execute_input":"2024-02-09T05:55:08.860004Z","iopub.status.idle":"2024-02-09T05:55:08.871250Z","shell.execute_reply.started":"2024-02-09T05:55:08.859972Z","shell.execute_reply":"2024-02-09T05:55:08.869946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-09T05:55:09.991343Z","iopub.execute_input":"2024-02-09T05:55:09.991952Z","iopub.status.idle":"2024-02-09T05:55:09.998832Z","shell.execute_reply.started":"2024-02-09T05:55:09.991917Z","shell.execute_reply":"2024-02-09T05:55:09.997746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### submission","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred.mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T16:39:51.138868Z","iopub.execute_input":"2024-02-08T16:39:51.139363Z","iopub.status.idle":"2024-02-08T16:39:51.178632Z","shell.execute_reply.started":"2024-02-08T16:39:51.139331Z","shell.execute_reply":"2024-02-08T16:39:51.177132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-02-08T16:39:51.181668Z","iopub.execute_input":"2024-02-08T16:39:51.182044Z","iopub.status.idle":"2024-02-08T16:39:51.190739Z","shell.execute_reply.started":"2024-02-08T16:39:51.182011Z","shell.execute_reply":"2024-02-08T16:39:51.189750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-08T16:39:51.208638Z","iopub.execute_input":"2024-02-08T16:39:51.209979Z","iopub.status.idle":"2024-02-08T16:39:51.229560Z","shell.execute_reply.started":"2024-02-08T16:39:51.209943Z","shell.execute_reply":"2024-02-08T16:39:51.228373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-08T16:39:51.232163Z","iopub.execute_input":"2024-02-08T16:39:51.232534Z","iopub.status.idle":"2024-02-08T16:39:51.246816Z","shell.execute_reply.started":"2024-02-08T16:39:51.232503Z","shell.execute_reply":"2024-02-08T16:39:51.245598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}