{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The purpose of this project is that consumer finance providers must accurately determine which clients can repay a loan.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-29T15:05:58.831039Z","iopub.execute_input":"2024-04-29T15:05:58.832896Z","iopub.status.idle":"2024-04-29T15:05:58.870776Z","shell.execute_reply.started":"2024-04-29T15:05:58.832818Z","shell.execute_reply":"2024-04-29T15:05:58.869913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl \nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:58.872768Z","iopub.execute_input":"2024-04-29T15:05:58.873593Z","iopub.status.idle":"2024-04-29T15:05:58.880643Z","shell.execute_reply.started":"2024-04-29T15:05:58.873564Z","shell.execute_reply":"2024-04-29T15:05:58.879716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- More than 45 .csv files\n\n- That contains information such as: Case Id, education level of the clients, marital status, situation of employee, house type, date decision, week number, credit balance, deposit balance, Tax deductions amount, amount of incoming, amount of outgoing, monthly annuity amount, card blocking reason, number of children, contract status, amount of down payment, ... ","metadata":{}},{"cell_type":"code","source":"direction= \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:58.881964Z","iopub.execute_input":"2024-04-29T15:05:58.882540Z","iopub.status.idle":"2024-04-29T15:05:58.893011Z","shell.execute_reply.started":"2024-04-29T15:05:58.882512Z","shell.execute_reply":"2024-04-29T15:05:58.892143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Various predictors were transformed, therefore we have the following notation for similar groups of transformations\n\nP - Transform DPD (Days past due)\n\nM - Masking categories\n\nA - Transform amount\n\nD - Transform date\n\nT - Unspecified Transform\n\nL - Unspecified Transform\n","metadata":{}},{"cell_type":"code","source":"def table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n # Cast Transform DPD (Days past due, P) and Transform Amount (A) as Float64\n    for col in df.columns:\n\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:58.894498Z","iopub.execute_input":"2024-04-29T15:05:58.895095Z","iopub.status.idle":"2024-04-29T15:05:58.904877Z","shell.execute_reply.started":"2024-04-29T15:05:58.895064Z","shell.execute_reply":"2024-04-29T15:05:58.904060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n\n    for col in df.columns:\n\n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:58.907126Z","iopub.execute_input":"2024-04-29T15:05:58.907732Z","iopub.status.idle":"2024-04-29T15:05:58.920593Z","shell.execute_reply.started":"2024-04-29T15:05:58.907704Z","shell.execute_reply":"2024-04-29T15:05:58.919747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_null(df: pl.DataFrame):\n    for col in df.columns:\n        df = df.with_columns(\n          pl.col(col).fill_null(pl.lit(df[col].mode()))\n      )\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:58.924774Z","iopub.execute_input":"2024-04-29T15:05:58.925317Z","iopub.status.idle":"2024-04-29T15:05:58.930917Z","shell.execute_reply.started":"2024-04-29T15:05:58.925282Z","shell.execute_reply":"2024-04-29T15:05:58.930006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Train dataset**","metadata":{}},{"cell_type":"code","source":"train_basetable = pl.read_csv(direction + \"csv_files/train/train_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:58.932349Z","iopub.execute_input":"2024-04-29T15:05:58.932652Z","iopub.status.idle":"2024-04-29T15:05:59.287206Z","shell.execute_reply.started":"2024-04-29T15:05:58.932629Z","shell.execute_reply":"2024-04-29T15:05:59.286272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static = pl.concat([pl.read_csv(direction + \"csv_files/train/train_static_0_0.csv\").pipe(table_dtypes),\n               pl.read_csv(direction + \"csv_files/train/train_static_0_1.csv\").pipe(table_dtypes),],how=\"vertical_relaxed\",)\ntrain_static.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:05:59.292218Z","iopub.execute_input":"2024-04-29T15:05:59.295062Z","iopub.status.idle":"2024-04-29T15:06:12.037905Z","shell.execute_reply.started":"2024-04-29T15:05:59.295022Z","shell.execute_reply":"2024-04-29T15:06:12.036726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static_cb = pl.read_csv(direction + \"csv_files/train/train_static_cb_0.csv\").pipe(table_dtypes)\ntrain_static_cb.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:12.040202Z","iopub.execute_input":"2024-04-29T15:06:12.040702Z","iopub.status.idle":"2024-04-29T15:06:13.688377Z","shell.execute_reply.started":"2024-04-29T15:06:12.040663Z","shell.execute_reply":"2024-04-29T15:06:13.687064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_person_1 = pl.read_csv(direction + \"csv_files/train/train_person_1.csv\").pipe(table_dtypes)\ntrain_person_1.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:13.693756Z","iopub.execute_input":"2024-04-29T15:06:13.694472Z","iopub.status.idle":"2024-04-29T15:06:16.870035Z","shell.execute_reply.started":"2024-04-29T15:06:13.694433Z","shell.execute_reply":"2024-04-29T15:06:16.868990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_credit_bureau_b_2 = pl.read_csv(direction + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:16.871614Z","iopub.execute_input":"2024-04-29T15:06:16.871940Z","iopub.status.idle":"2024-04-29T15:06:17.071783Z","shell.execute_reply.started":"2024-04-29T15:06:16.871911Z","shell.execute_reply":"2024-04-29T15:06:17.070899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Test dataset**","metadata":{}},{"cell_type":"code","source":"test_basetable = pl.read_csv(direction + \"csv_files/test/test_base.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:17.073368Z","iopub.execute_input":"2024-04-29T15:06:17.074005Z","iopub.status.idle":"2024-04-29T15:06:17.079943Z","shell.execute_reply.started":"2024-04-29T15:06:17.073974Z","shell.execute_reply":"2024-04-29T15:06:17.079155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_static = pl.concat([pl.read_csv(direction + \"csv_files/test/test_static_0_0.csv\").pipe(table_dtypes),\n                         pl.read_csv(direction + \"csv_files/test/test_static_0_1.csv\").pipe(table_dtypes),\n                         pl.read_csv(direction + \"csv_files/test/test_static_0_2.csv\").pipe(table_dtypes),],how=\"vertical_relaxed\",)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:17.081332Z","iopub.execute_input":"2024-04-29T15:06:17.081955Z","iopub.status.idle":"2024-04-29T15:06:17.111508Z","shell.execute_reply.started":"2024-04-29T15:06:17.081925Z","shell.execute_reply":"2024-04-29T15:06:17.110550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_static_cb = pl.read_csv(direction + \"csv_files/test/test_static_cb_0.csv\").pipe(table_dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:17.113080Z","iopub.execute_input":"2024-04-29T15:06:17.113748Z","iopub.status.idle":"2024-04-29T15:06:17.120151Z","shell.execute_reply.started":"2024-04-29T15:06:17.113717Z","shell.execute_reply":"2024-04-29T15:06:17.119133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1 = pl.read_csv(direction + \"csv_files/test/test_person_1.csv\").pipe(table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:17.121692Z","iopub.execute_input":"2024-04-29T15:06:17.121990Z","iopub.status.idle":"2024-04-29T15:06:17.130310Z","shell.execute_reply.started":"2024-04-29T15:06:17.121965Z","shell.execute_reply":"2024-04-29T15:06:17.129382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_credit_bureau_b_2 = pl.read_csv(direction + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:17.131728Z","iopub.execute_input":"2024-04-29T15:06:17.132026Z","iopub.status.idle":"2024-04-29T15:06:17.142613Z","shell.execute_reply.started":"2024-04-29T15:06:17.132001Z","shell.execute_reply":"2024-04-29T15:06:17.141200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#main occupation # income \ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:17.144540Z","iopub.execute_input":"2024-04-29T15:06:17.144880Z","iopub.status.idle":"2024-04-29T15:06:18.344057Z","shell.execute_reply.started":"2024-04-29T15:06:17.144852Z","shell.execute_reply":"2024-04-29T15:06:18.343200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"P - Transform DPD (Days past due)\n\nM - Masking categories\n\nA - Transform amount\n\nD - Transform date\n\nT - Unspecified Transform\n\nL - Unspecified Transform","metadata":{}},{"cell_type":"code","source":"# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:18.345431Z","iopub.execute_input":"2024-04-29T15:06:18.346085Z","iopub.status.idle":"2024-04-29T15:06:18.355914Z","shell.execute_reply.started":"2024-04-29T15:06:18.346054Z","shell.execute_reply":"2024-04-29T15:06:18.350479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#consumer finance A-type M-type   (capabilities) #pmt --- payment  #marital\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:18.357530Z","iopub.execute_input":"2024-04-29T15:06:18.357852Z","iopub.status.idle":"2024-04-29T15:06:18.365284Z","shell.execute_reply.started":"2024-04-29T15:06:18.357826Z","shell.execute_reply":"2024-04-29T15:06:18.364164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:18.366927Z","iopub.execute_input":"2024-04-29T15:06:18.367269Z","iopub.status.idle":"2024-04-29T15:06:19.706915Z","shell.execute_reply.started":"2024-04-29T15:06:18.367227Z","shell.execute_reply":"2024-04-29T15:06:19.705739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\ndata_test = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:19.708470Z","iopub.execute_input":"2024-04-29T15:06:19.708792Z","iopub.status.idle":"2024-04-29T15:06:19.721145Z","shell.execute_reply.started":"2024-04-29T15:06:19.708764Z","shell.execute_reply":"2024-04-29T15:06:19.720344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- case_id - This is the unique identifier for each credit case. You'll need this ID to join relevant tables to the base table.\n- date_decision - This refers to the date when a decision was made regarding the approval of the loan.\n- WEEK_NUM - This is the week number used for aggregation. In the test sample, WEEK_NUM continues sequentially from the last training value of WEEK_NUM.\n- MONTH - This column represents the month and is intended for aggregation purposes.\n- target - This is the target value, determined after a certain period based on whether or not the client defaulted on the specific credit case (loan).\n- num_group1 - This is an indexing column used for the historical records of case_id in both depth=1 and depth=2 tables.\n- num_group2 - This is the second indexing column for depth=2 tables' historical records of case_id. The order of num_group1 and num_group2 is important and will be clarified in feature definitions.","metadata":{}},{"cell_type":"code","source":"case_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.6, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\nprint(cols_pred)\n\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:19.722384Z","iopub.execute_input":"2024-04-29T15:06:19.722868Z","iopub.status.idle":"2024-04-29T15:06:25.735084Z","shell.execute_reply.started":"2024-04-29T15:06:19.722842Z","shell.execute_reply":"2024-04-29T15:06:25.734182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:25.740782Z","iopub.execute_input":"2024-04-29T15:06:25.741368Z","iopub.status.idle":"2024-04-29T15:06:25.745795Z","shell.execute_reply.started":"2024-04-29T15:06:25.741337Z","shell.execute_reply":"2024-04-29T15:06:25.745035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Applied LightGBM (Light Gradient Boosting Machine) for efficient prediction: an open-source, distributed, high-performance gradient boosting framework developed by Microsoft. ","metadata":{}},{"cell_type":"code","source":"lgb_train = lgb.Dataset(X_train, label=y_train)\nlgb_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_train)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}\n\ngbm = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,\n    callbacks=[lgb.log_evaluation(100), lgb.early_stopping(20)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:06:25.746920Z","iopub.execute_input":"2024-04-29T15:06:25.747393Z","iopub.status.idle":"2024-04-29T15:08:08.205289Z","shell.execute_reply.started":"2024-04-29T15:06:25.747366Z","shell.execute_reply":"2024-04-29T15:08:08.204491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    y_pred = gbm.predict(X, num_iteration=gbm.best_iteration)\n    base[\"score\"] = y_pred\n\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}') \nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}') \nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}')  ","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:08.206624Z","iopub.execute_input":"2024-04-29T15:08:08.207103Z","iopub.status.idle":"2024-04-29T15:08:35.842369Z","shell.execute_reply.started":"2024-04-29T15:08:08.207074Z","shell.execute_reply":"2024-04-29T15:08:35.841403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = pd.DataFrame({\n    'Feature': X_train.columns,\n    'Importance': gbm.feature_importance(importance_type='split')\n}).sort_values(by='Importance', ascending=False)\n\n# Plotting feature importances\nplt.figure(figsize=(10, 6))\nsns.barplot(data=feature_importances, x='Importance', y='Feature')\nplt.title('Feature Importance from LightGBM')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:35.843643Z","iopub.execute_input":"2024-04-29T15:08:35.844156Z","iopub.status.idle":"2024-04-29T15:08:36.606798Z","shell.execute_reply.started":"2024-04-29T15:08:35.844129Z","shell.execute_reply":"2024-04-29T15:08:36.605442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Area Under the ROC Curve (AUC) quantifies the overall performance of a classification model. It measures the area under the ROC curve, ranging from 0 to 1. score of 0.5 means the model performs no better than random guessing, while a score of 1.0 signifies that the model perfectly discriminates between positive and negative instances across all thresholds.","metadata":{}},{"cell_type":"markdown","source":"#Evaluation\n\nSubmissions are evaluated using a gini stability metric. A gini score is calculated for predictions corresponding to each WEEK_NUM.\n\ngini=2∗AUC−1\nA linear regression, a⋅x+b\n, is fit through the weekly gini scores, and a falling_rate is calculated as min(0,a)\n. This is used to penalize models that drop off in predictive ability.","metadata":{}},{"cell_type":"markdown","source":"#case_id - This is the unique identifier for each credit case. You'll need this ID to join relevant tables to the base table.\n\n#date_decision - This refers to the date when a decision was made regarding the approval of the loan.\n\n#WEEK_NUM - This is the week number used for aggregation. In the test sample, WEEK_NUM continues sequentially from the last training value of WEEK_NUM.","metadata":{}},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}') ","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:36.608702Z","iopub.execute_input":"2024-04-29T15:08:36.609147Z","iopub.status.idle":"2024-04-29T15:08:37.615956Z","shell.execute_reply.started":"2024-04-29T15:08:36.609107Z","shell.execute_reply":"2024-04-29T15:08:37.614604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming predictions are already made in 'base_train', 'base_valid', 'base_test'\n# Calculate and print AUC scores\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}')\nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}')\nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}')\n\n# Calculate and print accuracy scores\naccuracy_train = accuracy_score(base_train[\"target\"], base_train[\"score\"].round())\naccuracy_valid = accuracy_score(base_valid[\"target\"], base_valid[\"score\"].round())\naccuracy_test = accuracy_score(base_test[\"target\"], base_test[\"score\"].round())\n\nprint(f'The accuracy score on the train set is: {accuracy_train}')\nprint(f'The accuracy score on the valid set is: {accuracy_valid}')\nprint(f'The accuracy score on the test set is: {accuracy_test}')","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:37.617852Z","iopub.execute_input":"2024-04-29T15:08:37.618304Z","iopub.status.idle":"2024-04-29T15:08:38.352505Z","shell.execute_reply.started":"2024-04-29T15:08:37.618244Z","shell.execute_reply":"2024-04-29T15:08:38.351253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = X_train.select_dtypes(include=['category']).columns","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:38.353989Z","iopub.execute_input":"2024-04-29T15:08:38.354435Z","iopub.status.idle":"2024-04-29T15:08:38.363828Z","shell.execute_reply.started":"2024-04-29T15:08:38.354395Z","shell.execute_reply":"2024-04-29T15:08:38.362417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_submission = data_test[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)\ncategorical_cols = X_train.select_dtypes(include=['category']).columns\n\nfor col in categorical_cols:\n    train_categories = set(X_train[col].cat.categories)\n    submission_categories = set(X_submission[col].cat.categories)\n    new_categories = submission_categories - train_categories\n    X_submission.loc[X_submission[col].isin(new_categories), col] = \"Unknown\"\n    new_dtype = pd.CategoricalDtype(categories=train_categories, ordered=True)\n    X_train[col] = X_train[col].astype(new_dtype)\n    X_submission[col] = X_submission[col].astype(new_dtype)\n\ny_submission_pred = gbm.predict(X_submission, num_iteration=gbm.best_iteration)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:38.365942Z","iopub.execute_input":"2024-04-29T15:08:38.366412Z","iopub.status.idle":"2024-04-29T15:08:38.473779Z","shell.execute_reply.started":"2024-04-29T15:08:38.366371Z","shell.execute_reply":"2024-04-29T15:08:38.472917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_test[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:38.475202Z","iopub.execute_input":"2024-04-29T15:08:38.475636Z","iopub.status.idle":"2024-04-29T15:08:38.485322Z","shell.execute_reply.started":"2024-04-29T15:08:38.475600Z","shell.execute_reply":"2024-04-29T15:08:38.483990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:17:59.717345Z","iopub.execute_input":"2024-04-29T15:17:59.717754Z","iopub.status.idle":"2024-04-29T15:17:59.735780Z","shell.execute_reply.started":"2024-04-29T15:17:59.717724Z","shell.execute_reply":"2024-04-29T15:17:59.734350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split data**","metadata":{}},{"cell_type":"code","source":"df_train= data.to_pandas()\ndf_test= data_test.to_pandas ()\ncase_id = df_test['case_id']\nna_counts = df_test.isna().sum()\ncols_test  = na_counts[na_counts < 4].index\ndf_test  = df_test[cols_test]","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:38.486791Z","iopub.execute_input":"2024-04-29T15:08:38.487208Z","iopub.status.idle":"2024-04-29T15:08:39.898991Z","shell.execute_reply.started":"2024-04-29T15:08:38.487170Z","shell.execute_reply":"2024-04-29T15:08:39.898032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:39.900663Z","iopub.execute_input":"2024-04-29T15:08:39.901062Z","iopub.status.idle":"2024-04-29T15:08:41.062188Z","shell.execute_reply.started":"2024-04-29T15:08:39.901025Z","shell.execute_reply":"2024-04-29T15:08:41.060866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols  = df_train.columns\ncols = list(set(cols_test).intersection(train_cols))","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:41.064076Z","iopub.execute_input":"2024-04-29T15:08:41.064537Z","iopub.status.idle":"2024-04-29T15:08:41.070035Z","shell.execute_reply.started":"2024-04-29T15:08:41.064498Z","shell.execute_reply":"2024-04-29T15:08:41.068937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = df_train.target.copy()\ndf_train = df_train[cols]\ndf_train['target'] = target.copy()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:41.071833Z","iopub.execute_input":"2024-04-29T15:08:41.072278Z","iopub.status.idle":"2024-04-29T15:08:41.522765Z","shell.execute_reply.started":"2024-04-29T15:08:41.072221Z","shell.execute_reply":"2024-04-29T15:08:41.521749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dropna(inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:41.524066Z","iopub.execute_input":"2024-04-29T15:08:41.524405Z","iopub.status.idle":"2024-04-29T15:08:42.784078Z","shell.execute_reply.started":"2024-04-29T15:08:41.524376Z","shell.execute_reply":"2024-04-29T15:08:42.783149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:42.785386Z","iopub.execute_input":"2024-04-29T15:08:42.786545Z","iopub.status.idle":"2024-04-29T15:08:43.049538Z","shell.execute_reply.started":"2024-04-29T15:08:42.786507Z","shell.execute_reply":"2024-04-29T15:08:43.048368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categ_cols = df_train.select_dtypes(include=['object']).columns","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:43.051084Z","iopub.execute_input":"2024-04-29T15:08:43.052178Z","iopub.status.idle":"2024-04-29T15:08:43.126696Z","shell.execute_reply.started":"2024-04-29T15:08:43.052140Z","shell.execute_reply":"2024-04-29T15:08:43.125792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ordinal_encoder = OrdinalEncoder(handle_unknown='use_encoded_value',\n                                 unknown_value=-1)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:43.128074Z","iopub.execute_input":"2024-04-29T15:08:43.129316Z","iopub.status.idle":"2024-04-29T15:08:43.138900Z","shell.execute_reply.started":"2024-04-29T15:08:43.129272Z","shell.execute_reply":"2024-04-29T15:08:43.137701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting objcets to numeric \ndf_categ = ordinal_encoder.fit_transform(df_train[categ_cols])\ndf_categ  = pd.DataFrame( df_categ ) \ndf_categ.columns = categ_cols","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:43.140431Z","iopub.execute_input":"2024-04-29T15:08:43.140851Z","iopub.status.idle":"2024-04-29T15:08:44.442358Z","shell.execute_reply.started":"2024-04-29T15:08:43.140813Z","shell.execute_reply":"2024-04-29T15:08:44.441316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns  = categ_cols , inplace = True)\ndf_train  = pd.concat([df_train , df_categ] , axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.443854Z","iopub.execute_input":"2024-04-29T15:08:44.444335Z","iopub.status.idle":"2024-04-29T15:08:44.641674Z","shell.execute_reply.started":"2024-04-29T15:08:44.444223Z","shell.execute_reply":"2024-04-29T15:08:44.640546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dropna(inplace =  True)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.642914Z","iopub.execute_input":"2024-04-29T15:08:44.643362Z","iopub.status.idle":"2024-04-29T15:08:44.704932Z","shell.execute_reply.started":"2024-04-29T15:08:44.643315Z","shell.execute_reply":"2024-04-29T15:08:44.703976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.706218Z","iopub.execute_input":"2024-04-29T15:08:44.707177Z","iopub.status.idle":"2024-04-29T15:08:44.753482Z","shell.execute_reply.started":"2024-04-29T15:08:44.707138Z","shell.execute_reply":"2024-04-29T15:08:44.752320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['date_decision'] = df_train['date_decision'].astype(int) // 10**9\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.754863Z","iopub.execute_input":"2024-04-29T15:08:44.755923Z","iopub.status.idle":"2024-04-29T15:08:44.776680Z","shell.execute_reply.started":"2024-04-29T15:08:44.755885Z","shell.execute_reply":"2024-04-29T15:08:44.775486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols.remove('case_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.778140Z","iopub.execute_input":"2024-04-29T15:08:44.778578Z","iopub.status.idle":"2024-04-29T15:08:44.783535Z","shell.execute_reply.started":"2024-04-29T15:08:44.778541Z","shell.execute_reply":"2024-04-29T15:08:44.782340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X= df_train[cols]\ny= df_train.target\nX_train, X_test , y_train, y_test = train_test_split(X , y , test_size  = 0.30 , random_state =  0 )\nprint(X_train.shape)\nprint(X_test.shape)\nprint(y_train.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.784994Z","iopub.execute_input":"2024-04-29T15:08:44.785562Z","iopub.status.idle":"2024-04-29T15:08:44.813213Z","shell.execute_reply.started":"2024-04-29T15:08:44.785523Z","shell.execute_reply":"2024-04-29T15:08:44.812140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Linear Regression, Decision Trees, Random Forest, and Gradient Boosting Regressor","metadata":{}},{"cell_type":"code","source":"dtree = DecisionTreeClassifier(max_depth=3)  # You can adjust parameters as needed\ndtree.fit(X_train, y_train)\ny_pred_dtree = dtree.predict(X_test)\nacc_dtree = accuracy_score(y_test, y_pred_dtree)\nprint(f'Decision Tree Accuracy: {acc_dtree}')","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.814375Z","iopub.execute_input":"2024-04-29T15:08:44.814681Z","iopub.status.idle":"2024-04-29T15:08:44.982912Z","shell.execute_reply.started":"2024-04-29T15:08:44.814657Z","shell.execute_reply":"2024-04-29T15:08:44.981536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestClassifier(n_estimators=100, max_depth=3, random_state=1)\nrf.fit(X_train, y_train)\ny_pred_rf = rf.predict(X_test)\nacc_rf = accuracy_score(y_test, y_pred_rf)\nprint(f'Random Forest Accuracy: {acc_rf}')","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:44.984883Z","iopub.execute_input":"2024-04-29T15:08:44.985348Z","iopub.status.idle":"2024-04-29T15:08:46.941676Z","shell.execute_reply.started":"2024-04-29T15:08:44.985298Z","shell.execute_reply":"2024-04-29T15:08:46.940365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_train_dtree = dtree.predict(X_train)\ntrain_acc_dtree = accuracy_score(y_train, y_pred_train_dtree)\nprint(f'Training set accuracy for Decision Tree: {train_acc_dtree}')","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:46.943735Z","iopub.execute_input":"2024-04-29T15:08:46.944204Z","iopub.status.idle":"2024-04-29T15:08:46.960820Z","shell.execute_reply.started":"2024-04-29T15:08:46.944162Z","shell.execute_reply":"2024-04-29T15:08:46.959684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = pd.DataFrame({\n    'Feature': X_train.columns,\n    'Importance': rf.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\nplt.figure(figsize=(10, 6))\nsns.barplot(data=feature_importances, x='Importance', y='Feature')\nplt.title('Feature Importance from Random Forest')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:46.962612Z","iopub.execute_input":"2024-04-29T15:08:46.963041Z","iopub.status.idle":"2024-04-29T15:08:47.578520Z","shell.execute_reply.started":"2024-04-29T15:08:46.963003Z","shell.execute_reply":"2024-04-29T15:08:47.577462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = pd.DataFrame({\n    'Feature': X_train.columns,\n    'Decision Tree': dtree.feature_importances_,\n    'Random Forest': rf.feature_importances_\n})\n\n\nfeature_importances.set_index('Feature', inplace=True)\nfeature_importances_normalized = feature_importances.div(feature_importances.sum(axis=0), axis=1)\nfeature_importances_normalized.reset_index(inplace=True)\n\n\nmelted_importances = pd.melt(feature_importances_normalized, id_vars=[\"Feature\"], var_name=\"Model\", value_name=\"Importance\")\n\n\nplt.figure(figsize=(12, 8))\nsns.barplot(data=melted_importances, x='Importance', y='Feature', hue='Model')\nplt.title('Comparison of Feature Importance Across Models')\nplt.xlabel('Normalized Importance')\nplt.ylabel('Features')\nplt.legend(title='Model')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:47.579755Z","iopub.execute_input":"2024-04-29T15:08:47.580635Z","iopub.status.idle":"2024-04-29T15:08:48.481755Z","shell.execute_reply.started":"2024-04-29T15:08:47.580601Z","shell.execute_reply":"2024-04-29T15:08:48.472675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Employed other models includes Random Forest, Decision Tree, Logistics regression, SVM and K-NN","metadata":{}},{"cell_type":"code","source":"models = {\n    'Random Forest': RandomForestClassifier(n_estimators=100),\n    'Decision Tree': DecisionTreeClassifier(),\n    'Logistic Regression': LogisticRegression(max_iter=1000),\n    'SVM': SVC(probability=True),\n    'k-NN': KNeighborsClassifier(),\n    'Gradient Boosting': GradientBoostingClassifier()\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:48.483602Z","iopub.execute_input":"2024-04-29T15:08:48.483943Z","iopub.status.idle":"2024-04-29T15:08:48.490715Z","shell.execute_reply.started":"2024-04-29T15:08:48.483915Z","shell.execute_reply":"2024-04-29T15:08:48.489442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = {}\nfitted_models=[]\ncv_scores=[]\n\nfor name, model in models.items():\n\n    model.fit(X_train, y_train)\n    fitted_models.append(model)\n    \n    y_pred = model.predict(X_test)\n    y_prob = model.predict_proba(X_test)[:, 1] if hasattr(model, \"predict_proba\") else [0] * len(y_pred)\n    \n    accuracy = accuracy_score(y_test, y_pred)\n    auc = roc_auc_score(y_test, y_prob) if hasattr(model, \"predict_proba\") else 'N/A'\n    cv_scores.append(auc)\n    results[name] = {'Accuracy': accuracy, 'AUC': auc}\n\nresults_df = pd.DataFrame(results).T\nprint(results_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:08:48.492299Z","iopub.execute_input":"2024-04-29T15:08:48.493039Z","iopub.status.idle":"2024-04-29T15:10:12.040365Z","shell.execute_reply.started":"2024-04-29T15:08:48.492999Z","shell.execute_reply":"2024-04-29T15:10:12.038548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fitted_models","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:10:12.042035Z","iopub.execute_input":"2024-04-29T15:10:12.042384Z","iopub.status.idle":"2024-04-29T15:10:12.049623Z","shell.execute_reply.started":"2024-04-29T15:10:12.042355Z","shell.execute_reply":"2024-04-29T15:10:12.048782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n\nmodel = VotingModel(fitted_models)","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:10:12.050790Z","iopub.execute_input":"2024-04-29T15:10:12.051305Z","iopub.status.idle":"2024-04-29T15:10:12.062108Z","shell.execute_reply.started":"2024-04-29T15:10:12.051275Z","shell.execute_reply":"2024-04-29T15:10:12.061212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize and fit the voting model\nvoting_model = VotingModel(fitted_models)\nvoting_model.fit(X_train, y_train)\n\n# If you need to use predictions:\ny_pred_voting = voting_model.predict(X_test)\ny_prob_voting = voting_model.predict_proba(X_test) if hasattr(voting_model, 'predict_proba') else None\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:10:12.063319Z","iopub.execute_input":"2024-04-29T15:10:12.063819Z","iopub.status.idle":"2024-04-29T15:10:20.143675Z","shell.execute_reply.started":"2024-04-29T15:10:12.063790Z","shell.execute_reply":"2024-04-29T15:10:20.142724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_names = list(models.keys())\n\nfor i, model in enumerate(fitted_models):\n\n    if hasattr(model, 'feature_importances_'):\n        importances = model.feature_importances_\n        indices = np.argsort(importances)[::-1] \n\n        plt.figure(figsize=(10, 8))\n        plt.title(f'Feature Importances for {model_names[i]}')\n        plt.barh(range(len(indices)), importances[indices], color='b', align='center')\n        plt.yticks(range(len(indices)), [X_train.columns[j] for j in indices])\n        plt.xlabel('Relative Importance')\n        plt.gca().invert_yaxis()  \n        plt.show()\n    else:\n        print(f\"The model {model_names[i]} does not support feature importances.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-29T15:10:20.144976Z","iopub.execute_input":"2024-04-29T15:10:20.145555Z","iopub.status.idle":"2024-04-29T15:10:22.031610Z","shell.execute_reply.started":"2024-04-29T15:10:20.145526Z","shell.execute_reply":"2024-04-29T15:10:22.030643Z"},"trusted":true},"execution_count":null,"outputs":[]}]}