{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# setup","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\n# import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \nfrom catboost import CatBoostClassifier, Pool\nfrom tqdm import tqdm\nimport optuna\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:24.628530Z","iopub.execute_input":"2024-04-03T10:31:24.628878Z","iopub.status.idle":"2024-04-03T10:31:27.322615Z","shell.execute_reply.started":"2024-04-03T10:31:24.628850Z","shell.execute_reply":"2024-04-03T10:31:27.321351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# loading and preparing data ","metadata":{}},{"cell_type":"code","source":"defin = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:27.324742Z","iopub.execute_input":"2024-04-03T10:31:27.325267Z","iopub.status.idle":"2024-04-03T10:31:27.345463Z","shell.execute_reply.started":"2024-04-03T10:31:27.325234Z","shell.execute_reply":"2024-04-03T10:31:27.344417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defin","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:27.346969Z","iopub.execute_input":"2024-04-03T10:31:27.347601Z","iopub.status.idle":"2024-04-03T10:31:27.364257Z","shell.execute_reply.started":"2024-04-03T10:31:27.347559Z","shell.execute_reply":"2024-04-03T10:31:27.363410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:27.366071Z","iopub.execute_input":"2024-04-03T10:31:27.366356Z","iopub.status.idle":"2024-04-03T10:31:27.373855Z","shell.execute_reply.started":"2024-04-03T10:31:27.366333Z","shell.execute_reply":"2024-04-03T10:31:27.372940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:27.375027Z","iopub.execute_input":"2024-04-03T10:31:27.375359Z","iopub.status.idle":"2024-04-03T10:31:42.459927Z","shell.execute_reply.started":"2024-04-03T10:31:27.375329Z","shell.execute_reply":"2024-04-03T10:31:42.459141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:42.460971Z","iopub.execute_input":"2024-04-03T10:31:42.461254Z","iopub.status.idle":"2024-04-03T10:31:42.543000Z","shell.execute_reply.started":"2024-04-03T10:31:42.461207Z","shell.execute_reply":"2024-04-03T10:31:42.542281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:42.544057Z","iopub.execute_input":"2024-04-03T10:31:42.544369Z","iopub.status.idle":"2024-04-03T10:31:44.370597Z","shell.execute_reply.started":"2024-04-03T10:31:42.544345Z","shell.execute_reply":"2024-04-03T10:31:44.369727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:44.371685Z","iopub.execute_input":"2024-04-03T10:31:44.371940Z","iopub.status.idle":"2024-04-03T10:31:44.382969Z","shell.execute_reply.started":"2024-04-03T10:31:44.371918Z","shell.execute_reply":"2024-04-03T10:31:44.382150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"case_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\nprint(cols_pred)\n\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:44.384190Z","iopub.execute_input":"2024-04-03T10:31:44.384562Z","iopub.status.idle":"2024-04-03T10:31:51.391515Z","shell.execute_reply.started":"2024-04-03T10:31:44.384516Z","shell.execute_reply":"2024-04-03T10:31:51.390413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.395201Z","iopub.execute_input":"2024-04-03T10:31:51.395536Z","iopub.status.idle":"2024-04-03T10:31:51.400625Z","shell.execute_reply.started":"2024-04-03T10:31:51.395509Z","shell.execute_reply":"2024-04-03T10:31:51.399617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.401784Z","iopub.execute_input":"2024-04-03T10:31:51.402025Z","iopub.status.idle":"2024-04-03T10:31:51.432797Z","shell.execute_reply.started":"2024-04-03T10:31:51.402003Z","shell.execute_reply":"2024-04-03T10:31:51.431976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.433869Z","iopub.execute_input":"2024-04-03T10:31:51.434148Z","iopub.status.idle":"2024-04-03T10:31:51.532231Z","shell.execute_reply.started":"2024-04-03T10:31:51.434124Z","shell.execute_reply":"2024-04-03T10:31:51.531260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features_names = list(X_train.select_dtypes(include='float64').columns)\ncat_features = list(X_train.select_dtypes(include='category').columns)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.533434Z","iopub.execute_input":"2024-04-03T10:31:51.533738Z","iopub.status.idle":"2024-04-03T10:31:51.637009Z","shell.execute_reply.started":"2024-04-03T10:31:51.533714Z","shell.execute_reply":"2024-04-03T10:31:51.635998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in cat_features:\n    categories = X_train[feature].cat.categories.tolist()\n    if 'Unknown' not in categories:\n        X_train[feature] = X_train[feature].cat.add_categories(['Unknown'])\n        X_valid[feature] = X_valid[feature].cat.add_categories(['Unknown'])\n        X_test[feature] = X_test[feature].cat.add_categories(['Unknown'])\n\nX_train[cat_features] = X_train[cat_features].fillna('Unknown')\nX_valid[cat_features] = X_valid[cat_features].fillna('Unknown')\nX_test[cat_features] = X_test[cat_features].fillna('Unknown')","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.638499Z","iopub.execute_input":"2024-04-03T10:31:51.638803Z","iopub.status.idle":"2024-04-03T10:31:51.673907Z","shell.execute_reply.started":"2024-04-03T10:31:51.638777Z","shell.execute_reply":"2024-04-03T10:31:51.673050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# importance analysis","metadata":{}},{"cell_type":"code","source":"train_pool = Pool(X_train, y_train, cat_features=cat_features)\ntest_pool = Pool(X_test, y_test, cat_features=cat_features)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.675249Z","iopub.execute_input":"2024-04-03T10:31:51.675596Z","iopub.status.idle":"2024-04-03T10:31:51.834373Z","shell.execute_reply.started":"2024-04-03T10:31:51.675565Z","shell.execute_reply":"2024-04-03T10:31:51.833580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CatBoostClassifier(iterations=100, learning_rate=0.1, random_seed=42, l2_leaf_reg=0.5, task_type='GPU')","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.835612Z","iopub.execute_input":"2024-04-03T10:31:51.835956Z","iopub.status.idle":"2024-04-03T10:31:51.843538Z","shell.execute_reply.started":"2024-04-03T10:31:51.835926Z","shell.execute_reply":"2024-04-03T10:31:51.842498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_pool, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:31:51.844833Z","iopub.execute_input":"2024-04-03T10:31:51.845110Z","iopub.status.idle":"2024-04-03T10:32:23.183778Z","shell.execute_reply.started":"2024-04-03T10:31:51.845086Z","shell.execute_reply":"2024-04-03T10:32:23.182879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = model.get_feature_importance(test_pool)\n\nfor feature, importance in zip(range(48), feature_importances):\n    print(f\"Feature {feature}: {importance}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.185036Z","iopub.execute_input":"2024-04-03T10:32:23.185409Z","iopub.status.idle":"2024-04-03T10:32:23.586325Z","shell.execute_reply.started":"2024-04-03T10:32:23.185377Z","shell.execute_reply":"2024-04-03T10:32:23.585265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero_importance_features = [feature for feature, importance in enumerate(feature_importances) if importance == 0]\nprint(f\"zero importance features: {zero_importance_features}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.587617Z","iopub.execute_input":"2024-04-03T10:32:23.587964Z","iopub.status.idle":"2024-04-03T10:32:23.593432Z","shell.execute_reply.started":"2024-04-03T10:32:23.587934Z","shell.execute_reply":"2024-04-03T10:32:23.592509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.drop(X_train.columns[zero_importance_features], axis=1)\nX_valid = X_valid.drop(X_valid.columns[zero_importance_features], axis=1)\nX_test = X_test.drop(X_test.columns[zero_importance_features], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.594508Z","iopub.execute_input":"2024-04-03T10:32:23.595539Z","iopub.status.idle":"2024-04-03T10:32:23.730278Z","shell.execute_reply.started":"2024-04-03T10:32:23.595508Z","shell.execute_reply":"2024-04-03T10:32:23.729244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(X_train.columns))\nprint(len(X_valid.columns))\nprint(len(X_test.columns))","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.731475Z","iopub.execute_input":"2024-04-03T10:32:23.731766Z","iopub.status.idle":"2024-04-03T10:32:23.736836Z","shell.execute_reply.started":"2024-04-03T10:32:23.731742Z","shell.execute_reply":"2024-04-03T10:32:23.735896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features_names = list(X_train.select_dtypes(include='float64').columns)\ncat_features = list(X_train.select_dtypes(include='category').columns)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.738179Z","iopub.execute_input":"2024-04-03T10:32:23.738692Z","iopub.status.idle":"2024-04-03T10:32:23.842316Z","shell.execute_reply.started":"2024-04-03T10:32:23.738661Z","shell.execute_reply":"2024-04-03T10:32:23.841240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.843880Z","iopub.execute_input":"2024-04-03T10:32:23.844649Z","iopub.status.idle":"2024-04-03T10:32:23.934253Z","shell.execute_reply.started":"2024-04-03T10:32:23.844620Z","shell.execute_reply":"2024-04-03T10:32:23.933277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# stability metric","metadata":{}},{"cell_type":"code","source":"def stability_metric(y_true, y_pred):\n    gini = 2 * roc_auc_score(y_true, y_pred) - 1\n    residuals = y_true - y_pred\n    a = np.mean(residuals)\n    std_residuals = np.std(residuals)\n    \n    metric = np.mean(gini) + 88.0 * min(0, a) - 0.5 * std_residuals\n    return metric","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.935507Z","iopub.execute_input":"2024-04-03T10:32:23.935821Z","iopub.status.idle":"2024-04-03T10:32:23.941607Z","shell.execute_reply.started":"2024-04-03T10:32:23.935796Z","shell.execute_reply":"2024-04-03T10:32:23.940581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model learning","metadata":{}},{"cell_type":"code","source":"train_pool = Pool(X_train, y_train, cat_features=cat_features)\nvalid_pool = Pool(X_valid, y_valid, cat_features=cat_features)\ntest_pool = Pool(X_test, y_test, cat_features=cat_features)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:23.942652Z","iopub.execute_input":"2024-04-03T10:32:23.942996Z","iopub.status.idle":"2024-04-03T10:32:24.067726Z","shell.execute_reply.started":"2024-04-03T10:32:23.942963Z","shell.execute_reply":"2024-04-03T10:32:24.066721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        'iterations': trial.suggest_int('iterations', 100, 1000),\n        'learning_rate': trial.suggest_float('learning_rate', 1e-3, 1.0, log=True),\n        'depth': trial.suggest_int('depth', 2, 10),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1e-8, 100.0, log=True),\n        'random_strength': trial.suggest_float('random_strength', 1e-8, 10.0),\n        'bagging_temperature': trial.suggest_float('bagging_temperature', 0.01, 100.0, log=True),\n        'od_type': trial.suggest_categorical('od_type', ['IncToDec', 'Iter']),\n        'task_type': 'GPU'  \n    }\n    \n    model = CatBoostClassifier(**params)\n    model.fit(train_pool, eval_set=(valid_pool), verbose=False)\n    preds = model.predict_proba(valid_pool)[:, 1]\n    metric = stability_metric(y_valid, preds)\n    \n    return metric","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:24.069214Z","iopub.execute_input":"2024-04-03T10:32:24.069892Z","iopub.status.idle":"2024-04-03T10:32:24.077908Z","shell.execute_reply.started":"2024-04-03T10:32:24.069866Z","shell.execute_reply":"2024-04-03T10:32:24.076907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction='maximize')","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:24.080972Z","iopub.execute_input":"2024-04-03T10:32:24.081289Z","iopub.status.idle":"2024-04-03T10:32:24.090236Z","shell.execute_reply.started":"2024-04-03T10:32:24.081256Z","shell.execute_reply":"2024-04-03T10:32:24.089390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"runs = 1\n\nwith tqdm(total=runs) as pbar:\n    for i in range(runs):\n        study.optimize(objective, n_trials=1)\n        pbar.update(1)\n\nbest_params = study.best_params\nprint(\"best params:\", best_params)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:24.095621Z","iopub.execute_input":"2024-04-03T10:32:24.095894Z","iopub.status.idle":"2024-04-03T10:32:48.057957Z","shell.execute_reply.started":"2024-04-03T10:32:24.095871Z","shell.execute_reply":"2024-04-03T10:32:48.056876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = CatBoostClassifier(**best_params, task_type='GPU')\nbest_model.fit(train_pool, eval_set=(valid_pool), verbose=False)\n\npreds = best_model.predict_proba(test_pool)[:, 1]\nmetric = stability_metric(y_test, preds)\nprint(f\"stability_metric on test: {metric:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:32:48.059390Z","iopub.execute_input":"2024-04-03T10:32:48.059834Z","iopub.status.idle":"2024-04-03T10:33:11.989456Z","shell.execute_reply.started":"2024-04-03T10:32:48.059795Z","shell.execute_reply":"2024-04-03T10:33:11.988444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# submission","metadata":{}},{"cell_type":"code","source":"data_submission = data_submission.to_pandas()\n\ncase_ids = data_submission['case_id']\ndata_submission = data_submission[list(X_test.columns)]\n\nX_submission = data_submission","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:33:11.991047Z","iopub.execute_input":"2024-04-03T10:33:11.991829Z","iopub.status.idle":"2024-04-03T10:33:12.001443Z","shell.execute_reply.started":"2024-04-03T10:33:11.991794Z","shell.execute_reply":"2024-04-03T10:33:12.000596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submission","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:33:12.002707Z","iopub.execute_input":"2024-04-03T10:33:12.003262Z","iopub.status.idle":"2024-04-03T10:33:12.038266Z","shell.execute_reply.started":"2024-04-03T10:33:12.003213Z","shell.execute_reply":"2024-04-03T10:33:12.037288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submission_pred = best_model.predict_proba(X_submission,)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:33:12.039517Z","iopub.execute_input":"2024-04-03T10:33:12.039905Z","iopub.status.idle":"2024-04-03T10:33:12.049016Z","shell.execute_reply.started":"2024-04-03T10:33:12.039871Z","shell.execute_reply":"2024-04-03T10:33:12.047721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": case_ids.to_numpy(),\n    \"score\": y_submission_pred\n})\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:35:37.477701Z","iopub.execute_input":"2024-04-03T10:35:37.478085Z","iopub.status.idle":"2024-04-03T10:35:37.484860Z","shell.execute_reply.started":"2024-04-03T10:35:37.478055Z","shell.execute_reply":"2024-04-03T10:35:37.484031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:35:37.730353Z","iopub.execute_input":"2024-04-03T10:35:37.730633Z","iopub.status.idle":"2024-04-03T10:35:37.740181Z","shell.execute_reply.started":"2024-04-03T10:35:37.730610Z","shell.execute_reply":"2024-04-03T10:35:37.739189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:33:12.073859Z","iopub.execute_input":"2024-04-03T10:33:12.074194Z","iopub.status.idle":"2024-04-03T10:33:12.090385Z","shell.execute_reply.started":"2024-04-03T10:33:12.074162Z","shell.execute_reply":"2024-04-03T10:33:12.089364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_","metadata":{"execution":{"iopub.status.busy":"2024-04-03T10:33:12.091258Z","iopub.execute_input":"2024-04-03T10:33:12.091521Z","iopub.status.idle":"2024-04-03T10:33:12.104720Z","shell.execute_reply.started":"2024-04-03T10:33:12.091499Z","shell.execute_reply":"2024-04-03T10:33:12.103813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}