{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\nimport optuna\nfrom sklearn import metrics\nfrom catboost import CatBoostClassifier\nimport glob\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-17T21:06:58.460418Z","iopub.execute_input":"2024-03-17T21:06:58.460820Z","iopub.status.idle":"2024-03-17T21:06:58.466892Z","shell.execute_reply.started":"2024-03-17T21:06:58.460788Z","shell.execute_reply":"2024-03-17T21:06:58.465766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Version 3 is the same as version 2. Just added headers to the file.","metadata":{}},{"cell_type":"markdown","source":"# Data preparation","metadata":{}},{"cell_type":"code","source":"def preprocessor(mode='train'):\n    feats = pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_base.csv\")\n        \n    deposit = pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_deposit_1.csv\")\n    \n    for idx in range(1, len(deposit.columns)):\n        col = deposit.columns[idx]\n        column_type = deposit[col].dtype\n        is_numeric = (column_type == pl.datatypes.Int64) or (column_type == pl.datatypes.Float64) \n        if is_numeric:\n            feat = deposit.group_by('case_id').agg(pl.max(col).alias(f\"max_deposit_{col}\"),\n                                           pl.mean(col).alias(f\"mean_deposit_{col}\"),\n                                           pl.median(col).alias(f\"median_deposit_{col}\"),\n                                           pl.std(col).alias(f\"std_deposit_{col}\"),\n                                           pl.min(col).alias(f\"min_deposit_{col}\"),\n                                           pl.count(col).alias(f\"count_deposit_{col}\"),\n                                           pl.sum(col).alias(f\"sum_deposit_{col}\"),\n                                           pl.n_unique(col).alias(f\"n_unique_deposit_{col}\"),\n                                           pl.first(col).alias(f\"first_deposit_{col}\"),\n                                           pl.last(col).alias(f\"last_deposit_{col}\")\n                                         )\n            feats = feats.join(feat, on='case_id', how='left')\n\n    static_cb = pl.read_csv(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/{mode}/{mode}_static_cb_0.csv\")\n    feats = feats.join(static_cb,on='case_id',how='left')\n    del static_cb\n    gc.collect()\n\n    return feats","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:20:38.018510Z","iopub.execute_input":"2024-03-17T21:20:38.018912Z","iopub.status.idle":"2024-03-17T21:20:38.029516Z","shell.execute_reply.started":"2024-03-17T21:20:38.018879Z","shell.execute_reply":"2024-03-17T21:20:38.027798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_feats = preprocessor(mode='train')\ntrain_feats = train_feats.filter(train_feats['WEEK_NUM']>=40)\nprint(f\"len(train_feats):{len(train_feats)}\")\n\ntest_feats = preprocessor(mode='test')\nprint(f\"len(test_feats):{len(test_feats)}\")\n\nfor col in test_feats.columns:\n    if (train_feats[col].dtype == pl.datatypes.Float64) or (test_feats[col].dtype == pl.datatypes.Float64):\n        train_feats.with_columns(train_feats[col].cast(pl.datatypes.Float64))\n        test_feats.with_columns(test_feats[col].cast(pl.datatypes.Float64))\ntrain_feats = train_feats.to_pandas()\n\nstatic = pd.concat([pd.read_csv(p) for p in glob.glob(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_*\")],)\ntrain_feats = train_feats.merge(static, how = \"left\", on = \"case_id\")\ntrain_feats['date_decision'] = pd.to_datetime(train_feats['date_decision']).dt.to_pydatetime\n\ntest_feats = test_feats.to_pandas()\nstatic = pd.concat([pd.read_csv(p) for p in glob.glob(f\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_*\")],)\ntest_feats = test_feats.merge(static, how = \"left\", on = \"case_id\")\ntest_feats['date_decision'] = pd.to_datetime(test_feats['date_decision']).dt.to_pydatetime\n\nfor col in test_feats.columns:\n    n_unique = train_feats[col].nunique()\n    if n_unique == 2 and train_feats[col].dtype == 'object':\n        unique = train_feats[col].unique()\n        train_feats[col] = (train_feats[col]==unique[0]).astype(int)\n        test_feats[col] = (test_feats[col]==unique[0]).astype(int)\n    elif (n_unique < 10) and train_feats[col].dtype == 'object':\n        unique = train_feats[col].unique()\n        for idx in range(len(unique)):\n            if unique[idx] == unique[idx]:\n                train_feats[col + \"_\" + str(idx)] = (train_feats[col] == unique[idx]).astype(int)\n                test_feats[col + \"_\" + str(idx)] = (test_feats[col] == unique[idx]).astype(int)\n        train_feats.drop([col], axis=1, inplace=True)\n        test_feats.drop([col], axis=1, inplace=True)\n\n\ndrop_cols=[]\nfor col in test_feats.columns:\n    if (train_feats[col].dtype == 'object') or (test_feats[col].dtype == 'object') \\\n        or (train_feats[col].nunique() == 1) or train_feats[col].isna().mean() > 0.95:\n        drop_cols += [col]\n\ndrop_cols += ['case_id','WEEK_NUM','MONTH']\ntrain_feats = train_feats.drop(drop_cols, axis=1)\ntest_feats = test_feats.drop(drop_cols, axis=1)\n\ntrain_feats.fillna(-1, inplace=True)\ntest_feats.fillna(-1, inplace=True)\n\ntrain_feats.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:23:59.088623Z","iopub.execute_input":"2024-03-17T21:23:59.089099Z","iopub.status.idle":"2024-03-17T21:25:15.691095Z","shell.execute_reply.started":"2024-03-17T21:23:59.089066Z","shell.execute_reply":"2024-03-17T21:25:15.689673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameters Tuning with Optuna and CatBoost","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_target = train_feats['target']\ntrain_feats = train_feats.drop(columns = ['target'])\n\nX_train, X_valid, y_train, y_valid = train_test_split(train_feats, train_target,test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:25:15.693424Z","iopub.execute_input":"2024-03-17T21:25:15.693725Z","iopub.status.idle":"2024-03-17T21:25:17.793214Z","shell.execute_reply.started":"2024-03-17T21:25:15.693688Z","shell.execute_reply":"2024-03-17T21:25:17.792411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n    learning_rate = trial.suggest_float(\n        \"learning_rate\", 1e-4, 1, log=True\n    )\n    depth = trial.suggest_int(\n        \"depth\", 2, 10, step=1\n    )\n    iterations = trial.suggest_int(\n        \"iterations\", 500, 1000, step=50\n    )\n    l2_leaf_reg = trial.suggest_float(\n        \"l2_leaf_reg\", 1e-3, 10, log=True\n    )\n    \n    \n    model = CatBoostClassifier(\n        learning_rate=learning_rate, \n        depth=depth,\n        iterations=iterations,\n        l2_leaf_reg=l2_leaf_reg\n    )\n    model.fit(X_train, y_train)\n    probas = model.predict_proba(X_valid)\n    auc = metrics.roc_auc_score(y_valid, probas[:, 1])\n\n    return auc\n\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=10)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:27:37.788617Z","iopub.execute_input":"2024-03-17T21:27:37.789566Z","iopub.status.idle":"2024-03-17T21:30:24.627978Z","shell.execute_reply.started":"2024-03-17T21:27:37.789531Z","shell.execute_reply":"2024-03-17T21:30:24.626978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CatBoostClassifier(\n    learning_rate=study.best_params['learning_rate'],\n    depth=study.best_params['depth'],\n    iterations=study.best_params['iterations'],\n    l2_leaf_reg=study.best_params['l2_leaf_reg'],\n)\n\nmodel.fit(train_feats, train_target)\npreds_proba = model.predict_proba(test_feats)\nsubmission = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nsubmission['score'] = preds_proba[:, 1]\nsubmission.to_csv(\"submission.csv\", index=None)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:30:24.629538Z","iopub.execute_input":"2024-03-17T21:30:24.629833Z","iopub.status.idle":"2024-03-17T21:30:34.302722Z","shell.execute_reply.started":"2024-03-17T21:30:24.629807Z","shell.execute_reply":"2024-03-17T21:30:34.301504Z"},"trusted":true},"execution_count":null,"outputs":[]}]}