{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:14:39.956529Z","iopub.execute_input":"2024-05-28T02:14:39.956899Z","iopub.status.idle":"2024-05-28T02:14:41.822021Z","shell.execute_reply.started":"2024-05-28T02:14:39.956870Z","shell.execute_reply":"2024-05-28T02:14:41.820800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:14:41.824526Z","iopub.execute_input":"2024-05-28T02:14:41.824982Z","iopub.status.idle":"2024-05-28T02:14:41.834405Z","shell.execute_reply.started":"2024-05-28T02:14:41.824940Z","shell.execute_reply":"2024-05-28T02:14:41.833612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:14:41.835649Z","iopub.execute_input":"2024-05-28T02:14:41.836472Z","iopub.status.idle":"2024-05-28T02:14:59.668129Z","shell.execute_reply.started":"2024-05-28T02:14:41.836439Z","shell.execute_reply":"2024-05-28T02:14:59.666931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:14:59.670346Z","iopub.execute_input":"2024-05-28T02:14:59.670719Z","iopub.status.idle":"2024-05-28T02:14:59.690224Z","shell.execute_reply.started":"2024-05-28T02:14:59.670688Z","shell.execute_reply":"2024-05-28T02:14:59.689067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) ","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:14:59.691718Z","iopub.execute_input":"2024-05-28T02:14:59.692290Z","iopub.status.idle":"2024-05-28T02:14:59.781283Z","shell.execute_reply.started":"2024-05-28T02:14:59.692257Z","shell.execute_reply":"2024-05-28T02:14:59.780291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering\n\nIn this part, we can see a simple example of joining tables via `case_id`. Here the loading and joining is done with polars library. Polars library is blazingly fast and has much smaller memory footprint than pandas. ","metadata":{}},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:14:59.784387Z","iopub.execute_input":"2024-05-28T02:14:59.785076Z","iopub.status.idle":"2024-05-28T02:15:00.651230Z","shell.execute_reply.started":"2024-05-28T02:14:59.785042Z","shell.execute_reply":"2024-05-28T02:15:00.650081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:00.653422Z","iopub.execute_input":"2024-05-28T02:15:00.653889Z","iopub.status.idle":"2024-05-28T02:15:00.667093Z","shell.execute_reply.started":"2024-05-28T02:15:00.653836Z","shell.execute_reply":"2024-05-28T02:15:00.665597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:00.671265Z","iopub.execute_input":"2024-05-28T02:15:00.672542Z","iopub.status.idle":"2024-05-28T02:15:00.677261Z","shell.execute_reply.started":"2024-05-28T02:15:00.672478Z","shell.execute_reply":"2024-05-28T02:15:00.676244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:00.678883Z","iopub.execute_input":"2024-05-28T02:15:00.679307Z","iopub.status.idle":"2024-05-28T02:15:00.687786Z","shell.execute_reply.started":"2024-05-28T02:15:00.679267Z","shell.execute_reply":"2024-05-28T02:15:00.686813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport pandas as pd\n\n# Assuming 'data' is your Polar DataFrame\n# Convert to Pandas DataFrame\ndf = data.to_pandas()\n\n# Create the count plot\nax = sns.countplot(x=\"target\", data=df)\n\n# Customize the plot as desired\nax.figure.set_size_inches(12, 6)\nax.set_title('count for loan status', fontsize=16)\nax.set_ylabel('count', fontsize=14)\nax.set_xlabel('loan status', fontsize=14)\n\nplt.xticks([0, 1], ['non_default', 'default'])\nplt.suptitle(\n    \"Fig. 1: Top: A bar plot of good versus bad loans.\",\n    fontweight=\"bold\",\n    horizontalalignment=\"right\",\n)\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:00.689355Z","iopub.execute_input":"2024-05-28T02:15:00.689801Z","iopub.status.idle":"2024-05-28T02:15:01.312640Z","shell.execute_reply.started":"2024-05-28T02:15:00.689749Z","shell.execute_reply":"2024-05-28T02:15:01.311117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:50.888723Z","iopub.execute_input":"2024-05-28T02:15:50.889145Z","iopub.status.idle":"2024-05-28T02:15:50.920343Z","shell.execute_reply.started":"2024-05-28T02:15:50.889110Z","shell.execute_reply":"2024-05-28T02:15:50.919220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:51.727536Z","iopub.execute_input":"2024-05-28T02:15:51.728450Z","iopub.status.idle":"2024-05-28T02:15:51.751474Z","shell.execute_reply.started":"2024-05-28T02:15:51.728391Z","shell.execute_reply":"2024-05-28T02:15:51.750307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df['mainoccupationinc_384A_any_selfemployed']\ndel df['person_housetype']\ndel df['pmts_pmtsoverdue_635A_max']\ndel df['pmts_dpdvalue_108P_over31']","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:52.954656Z","iopub.execute_input":"2024-05-28T02:15:52.955048Z","iopub.status.idle":"2024-05-28T02:15:52.962073Z","shell.execute_reply.started":"2024-05-28T02:15:52.955018Z","shell.execute_reply":"2024-05-28T02:15:52.960931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['target']","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:54.202369Z","iopub.execute_input":"2024-05-28T02:15:54.202784Z","iopub.status.idle":"2024-05-28T02:15:54.212451Z","shell.execute_reply.started":"2024-05-28T02:15:54.202752Z","shell.execute_reply":"2024-05-28T02:15:54.211094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_default = df[df[\"target\"] == 1].copy()  \n\ndf_non_default = df[df[\"target\"] == 0].copy()\n\ntotal_default = df_default.shape[0]\ntotal_non_default = df_non_default.shape[0]\ntotal_loans = data.shape[0]\n\n\nprint('no of default cases:', total_default)\nprint('% of default cases : {:.2f}:', (total_default/total_loans)*100)\n\nprint('no of non_default cases:', total_non_default)\nprint('% of non-default cases : {:.2f}:', (total_non_default/total_loans)*100)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:55.296365Z","iopub.execute_input":"2024-05-28T02:15:55.298881Z","iopub.status.idle":"2024-05-28T02:15:55.520683Z","shell.execute_reply.started":"2024-05-28T02:15:55.298806Z","shell.execute_reply":"2024-05-28T02:15:55.519480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop('target', axis=1).values\ny = df['target'].values","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:56.503110Z","iopub.execute_input":"2024-05-28T02:15:56.504019Z","iopub.status.idle":"2024-05-28T02:15:56.954932Z","shell.execute_reply.started":"2024-05-28T02:15:56.503967Z","shell.execute_reply":"2024-05-28T02:15:56.954045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.1, random_state = 2022, stratify = y)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:57.868795Z","iopub.execute_input":"2024-05-28T02:15:57.869176Z","iopub.status.idle":"2024-05-28T02:15:59.659838Z","shell.execute_reply.started":"2024-05-28T02:15:57.869145Z","shell.execute_reply":"2024-05-28T02:15:59.658629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:59.661659Z","iopub.execute_input":"2024-05-28T02:15:59.662525Z","iopub.status.idle":"2024-05-28T02:15:59.669925Z","shell.execute_reply.started":"2024-05-28T02:15:59.662490Z","shell.execute_reply":"2024-05-28T02:15:59.668771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:15:59.934287Z","iopub.execute_input":"2024-05-28T02:15:59.934683Z","iopub.status.idle":"2024-05-28T02:15:59.941146Z","shell.execute_reply.started":"2024-05-28T02:15:59.934652Z","shell.execute_reply":"2024-05-28T02:15:59.940060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Model:","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from collections import Counter\n\n# EDA\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# data manipulation\nfrom scipy import stats\n\n\n# model evaluation\nfrom sklearn.metrics import (\n    accuracy_score,\n    brier_score_loss,\n    classification_report,\n    cohen_kappa_score,\n    f1_score,\n    precision_score,\n    recall_score,\n    roc_auc_score,\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:04.878935Z","iopub.execute_input":"2024-05-28T02:16:04.879387Z","iopub.status.idle":"2024-05-28T02:16:04.886080Z","shell.execute_reply.started":"2024-05-28T02:16:04.879351Z","shell.execute_reply":"2024-05-28T02:16:04.884531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# counts the number of classes before oversampling\nprint(\"Balancing of data:\", Counter(y_train))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:05.959249Z","iopub.execute_input":"2024-05-28T02:16:05.959678Z","iopub.status.idle":"2024-05-28T02:16:06.409970Z","shell.execute_reply.started":"2024-05-28T02:16:05.959643Z","shell.execute_reply":"2024-05-28T02:16:06.408861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.combine import SMOTETomek","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:07.573772Z","iopub.execute_input":"2024-05-28T02:16:07.574241Z","iopub.status.idle":"2024-05-28T02:16:08.013571Z","shell.execute_reply.started":"2024-05-28T02:16:07.574193Z","shell.execute_reply":"2024-05-28T02:16:08.012323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.DataFrame(X_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:09.070631Z","iopub.execute_input":"2024-05-28T02:16:09.071361Z","iopub.status.idle":"2024-05-28T02:16:09.077921Z","shell.execute_reply.started":"2024-05-28T02:16:09.071322Z","shell.execute_reply":"2024-05-28T02:16:09.076728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.DataFrame(X_test)\ny_train = pd.DataFrame(y_train)\ny_test = pd.DataFrame(y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:10.221317Z","iopub.execute_input":"2024-05-28T02:16:10.221764Z","iopub.status.idle":"2024-05-28T02:16:10.228308Z","shell.execute_reply.started":"2024-05-28T02:16:10.221728Z","shell.execute_reply":"2024-05-28T02:16:10.227345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.select_dtypes(include=[object])","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:11.463688Z","iopub.execute_input":"2024-05-28T02:16:11.464074Z","iopub.status.idle":"2024-05-28T02:16:11.631020Z","shell.execute_reply.started":"2024-05-28T02:16:11.464044Z","shell.execute_reply":"2024-05-28T02:16:11.629736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" del X_train[1]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:12.688478Z","iopub.execute_input":"2024-05-28T02:16:12.688909Z","iopub.status.idle":"2024-05-28T02:16:12.693723Z","shell.execute_reply.started":"2024-05-28T02:16:12.688871Z","shell.execute_reply":"2024-05-28T02:16:12.692888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:13.988056Z","iopub.execute_input":"2024-05-28T02:16:13.988973Z","iopub.status.idle":"2024-05-28T02:16:14.005500Z","shell.execute_reply.started":"2024-05-28T02:16:13.988931Z","shell.execute_reply":"2024-05-28T02:16:14.003962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"number_of_classes = len(set(y))\nnumber_of_classes","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:15.105267Z","iopub.execute_input":"2024-05-28T02:16:15.106394Z","iopub.status.idle":"2024-05-28T02:16:15.438252Z","shell.execute_reply.started":"2024-05-28T02:16:15.106352Z","shell.execute_reply":"2024-05-28T02:16:15.437452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest Classifier\n\n","metadata":{}},{"cell_type":"code","source":"# Set the threshold for defaults\nTHRESHOLD = 0.50","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:18.021423Z","iopub.execute_input":"2024-05-28T02:16:18.021844Z","iopub.status.idle":"2024-05-28T02:16:18.027157Z","shell.execute_reply.started":"2024-05-28T02:16:18.021811Z","shell.execute_reply":"2024-05-28T02:16:18.025933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:19.588370Z","iopub.execute_input":"2024-05-28T02:16:19.589627Z","iopub.status.idle":"2024-05-28T02:16:19.593992Z","shell.execute_reply.started":"2024-05-28T02:16:19.589577Z","shell.execute_reply":"2024-05-28T02:16:19.593134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier = RandomForestClassifier(random_state=2022),\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:21.985391Z","iopub.execute_input":"2024-05-28T02:16:21.986571Z","iopub.status.idle":"2024-05-28T02:16:21.992191Z","shell.execute_reply.started":"2024-05-28T02:16:21.986519Z","shell.execute_reply":"2024-05-28T02:16:21.990810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_model_metrics(model, X_test, y_test, model_probs, threshold):\n    \"\"\"\n    Calculates Accuracy, F1-Score, PR AUC\n    \"\"\"\n    # keeps probabilities for the positive outcome only\n    probs = pd.DataFrame(model_probs[:, 1], columns=[\"prob\"])\n\n    # applies the threshold\n    y_pred = probs[\"prob\"].apply(lambda x: 1 if x > threshold else 0)\n\n    # calculates f1-score\n    f1 = f1_score(y_test, y_pred)\n\n    # calculates accuracy\n    accuracy = accuracy_score(y_test, y_pred)\n\n    # calculates kappa score\n    kappa = cohen_kappa_score(y_test, y_pred)\n\n    # calculates AUC\n    auc_score = roc_auc_score(y_test, probs)\n\n    # calculates the precision\n    precision = precision_score(y_test, y_pred)\n\n    # calculates the recall\n    recall = recall_score(y_test, y_pred)\n\n    return accuracy, kappa, f1, auc_score, precision, recall","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:23.306533Z","iopub.execute_input":"2024-05-28T02:16:23.306929Z","iopub.status.idle":"2024-05-28T02:16:23.315807Z","shell.execute_reply.started":"2024-05-28T02:16:23.306898Z","shell.execute_reply":"2024-05-28T02:16:23.314792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_classifiers_performance(\n    X_train, X_test, y_train, y_test, threshold, classifiers\n):\n    # creates empty data frame\n    df_performance = pd.DataFrame()\n\n    for clf in classifiers:\n        print(\"Training \" + type(clf).__name__ + \"...\")\n        # fits the classifier to training data\n        clf.fit(X_train, y_train)\n\n        # predict the probabilities\n        clf_probs = clf.predict_proba(X_test)\n\n        # calculates model metrics\n        (\n            clf_accuracy,\n            clf_kappa,\n            clf_f1,\n            clf_auc,\n            clf_precision,\n            clf_recall,\n        ) = calculate_model_metrics(clf, X_test, y_test, clf_probs, threshold)\n\n        # creates a dict\n        clf_dict = {\n            \"model\": [type(clf).__name__, \"---\"],\n            \"precision\": [clf_precision, np.nan],\n            \"recall\": [clf_recall, np.nan],\n            \"f1-Score\": [clf_f1, np.nan],\n            \"ROC AUC\": [clf_auc, np.nan],\n            \"accuracy\": [clf_accuracy, np.nan],\n            \"cohen kappa\": [clf_kappa, np.nan],\n        }\n\n        # concatenate Data Frames\n        df_performance = pd.concat([df_performance, pd.DataFrame(clf_dict)])\n\n    # resets Data Frame index\n    df_performance = df_performance.reset_index()\n\n    # drops index\n    df_performance.drop(\"index\", axis=1, inplace=True)\n\n    # gets only the odd numbered rows\n    rows_to_drop = np.arange(1, len(classifiers) * 2, 2)\n\n    # drops unwanted rows that have no data\n    df_performance.drop(rows_to_drop, inplace=True)\n\n    # returns performance summary\n    return df_performance","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:24.340592Z","iopub.execute_input":"2024-05-28T02:16:24.340999Z","iopub.status.idle":"2024-05-28T02:16:24.354528Z","shell.execute_reply.started":"2024-05-28T02:16:24.340969Z","shell.execute_reply":"2024-05-28T02:16:24.353063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:25.560022Z","iopub.execute_input":"2024-05-28T02:16:25.560736Z","iopub.status.idle":"2024-05-28T02:16:25.566006Z","shell.execute_reply.started":"2024-05-28T02:16:25.560688Z","shell.execute_reply":"2024-05-28T02:16:25.565104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.to_numpy()\nX_test = X_test.to_numpy()\ny_test = y_test.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:26.637991Z","iopub.execute_input":"2024-05-28T02:16:26.638989Z","iopub.status.idle":"2024-05-28T02:16:26.948138Z","shell.execute_reply.started":"2024-05-28T02:16:26.638948Z","shell.execute_reply":"2024-05-28T02:16:26.946979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# Assuming y_train is a column vector\ny_train_flat = y_train.ravel()  # Reshape using ravel()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:28.137362Z","iopub.execute_input":"2024-05-28T02:16:28.137803Z","iopub.status.idle":"2024-05-28T02:16:28.143877Z","shell.execute_reply.started":"2024-05-28T02:16:28.137768Z","shell.execute_reply":"2024-05-28T02:16:28.142382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:29.543051Z","iopub.execute_input":"2024-05-28T02:16:29.543456Z","iopub.status.idle":"2024-05-28T02:16:29.551482Z","shell.execute_reply.started":"2024-05-28T02:16:29.543404Z","shell.execute_reply":"2024-05-28T02:16:29.550328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.DataFrame(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:30.860777Z","iopub.execute_input":"2024-05-28T02:16:30.861401Z","iopub.status.idle":"2024-05-28T02:16:30.866364Z","shell.execute_reply.started":"2024-05-28T02:16:30.861365Z","shell.execute_reply":"2024-05-28T02:16:30.865185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:32.233403Z","iopub.execute_input":"2024-05-28T02:16:32.234479Z","iopub.status.idle":"2024-05-28T02:16:32.249208Z","shell.execute_reply.started":"2024-05-28T02:16:32.234409Z","shell.execute_reply":"2024-05-28T02:16:32.248290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_test[1]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:33.369075Z","iopub.execute_input":"2024-05-28T02:16:33.370020Z","iopub.status.idle":"2024-05-28T02:16:33.374827Z","shell.execute_reply.started":"2024-05-28T02:16:33.369980Z","shell.execute_reply":"2024-05-28T02:16:33.373386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:35.318062Z","iopub.execute_input":"2024-05-28T02:16:35.319540Z","iopub.status.idle":"2024-05-28T02:16:35.338761Z","shell.execute_reply.started":"2024-05-28T02:16:35.319483Z","shell.execute_reply":"2024-05-28T02:16:35.337603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = X_test.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:36.381340Z","iopub.execute_input":"2024-05-28T02:16:36.382133Z","iopub.status.idle":"2024-05-28T02:16:36.420843Z","shell.execute_reply.started":"2024-05-28T02:16:36.382086Z","shell.execute_reply":"2024-05-28T02:16:36.419700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculates classifiers performance\ndf_performances = get_classifiers_performance(\n    X_train, X_test, y_train, y_test, THRESHOLD, classifier\n)\n# highlight max values for each column\ndf_performances.style.highlight_max()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:16:37.385623Z","iopub.execute_input":"2024-05-28T02:16:37.386381Z","iopub.status.idle":"2024-05-28T02:27:48.363804Z","shell.execute_reply.started":"2024-05-28T02:16:37.386329Z","shell.execute_reply":"2024-05-28T02:27:48.362518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Probability Distribution","metadata":{}},{"cell_type":"code","source":"# instantiates the classifiers\nrf_clf = RandomForestClassifier(random_state=2022)\n# trains the classifiers\nrf_clf.fit(X_train, y_train)\n\n# store the predicted probabilities for class 1\ny_pred_rf_prob = rf_clf.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T02:29:14.457777Z","iopub.execute_input":"2024-05-28T02:29:14.458239Z","iopub.status.idle":"2024-05-28T02:40:41.260751Z","shell.execute_reply.started":"2024-05-28T02:29:14.458206Z","shell.execute_reply":"2024-05-28T02:40:41.259333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:00:43.635760Z","iopub.execute_input":"2024-05-28T03:00:43.636169Z","iopub.status.idle":"2024-05-28T03:00:43.642190Z","shell.execute_reply.started":"2024-05-28T03:00:43.636138Z","shell.execute_reply":"2024-05-28T03:00:43.640781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = roc_auc_score(y_true, y_pred_rf_prob)\nscore","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:27:47.747803Z","iopub.execute_input":"2024-05-28T03:27:47.748228Z","iopub.status.idle":"2024-05-28T03:27:47.790066Z","shell.execute_reply.started":"2024-05-28T03:27:47.748196Z","shell.execute_reply":"2024-05-28T03:27:47.788871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"y_pred = rf_clf.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:24:10.262540Z","iopub.execute_input":"2024-05-28T03:24:10.262935Z","iopub.status.idle":"2024-05-28T03:24:17.613768Z","shell.execute_reply.started":"2024-05-28T03:24:10.262897Z","shell.execute_reply":"2024-05-28T03:24:17.612419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.DataFrame(y_train)\nX_train = pd.DataFrame(X_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:02:31.534920Z","iopub.execute_input":"2024-05-28T03:02:31.535363Z","iopub.status.idle":"2024-05-28T03:02:31.542731Z","shell.execute_reply.started":"2024-05-28T03:02:31.535328Z","shell.execute_reply":"2024-05-28T03:02:31.541327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:03:31.423266Z","iopub.execute_input":"2024-05-28T03:03:31.423742Z","iopub.status.idle":"2024-05-28T03:03:31.441869Z","shell.execute_reply.started":"2024-05-28T03:03:31.423705Z","shell.execute_reply":"2024-05-28T03:03:31.440595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:03:10.274051Z","iopub.execute_input":"2024-05-28T03:03:10.274495Z","iopub.status.idle":"2024-05-28T03:03:10.291346Z","shell.execute_reply.started":"2024-05-28T03:03:10.274460Z","shell.execute_reply":"2024-05-28T03:03:10.290150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:31:57.964474Z","iopub.execute_input":"2024-05-28T03:31:57.964957Z","iopub.status.idle":"2024-05-28T03:31:57.970794Z","shell.execute_reply.started":"2024-05-28T03:31:57.964915Z","shell.execute_reply":"2024-05-28T03:31:57.969534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\n\nScoring the submission dataset is below, we need to take care of new categories. Then we save the score as a last step. ","metadata":{}},{"cell_type":"code","source":"data_submission = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:31:59.742915Z","iopub.execute_input":"2024-05-28T03:31:59.743346Z","iopub.status.idle":"2024-05-28T03:31:59.754383Z","shell.execute_reply.started":"2024-05-28T03:31:59.743262Z","shell.execute_reply":"2024-05-28T03:31:59.753133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['case_id'] = data_submission['case_id']","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:32:00.600546Z","iopub.execute_input":"2024-05-28T03:32:00.600930Z","iopub.status.idle":"2024-05-28T03:32:00.608492Z","shell.execute_reply.started":"2024-05-28T03:32:00.600902Z","shell.execute_reply":"2024-05-28T03:32:00.607032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['score'] = score\nsubmission['score']","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:32:01.919285Z","iopub.execute_input":"2024-05-28T03:32:01.919705Z","iopub.status.idle":"2024-05-28T03:32:01.929358Z","shell.execute_reply.started":"2024-05-28T03:32:01.919673Z","shell.execute_reply":"2024-05-28T03:32:01.928123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_submission","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:32:04.099205Z","iopub.execute_input":"2024-05-28T03:32:04.099659Z","iopub.status.idle":"2024-05-28T03:32:04.111611Z","shell.execute_reply.started":"2024-05-28T03:32:04.099622Z","shell.execute_reply":"2024-05-28T03:32:04.110271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:32:06.257976Z","iopub.execute_input":"2024-05-28T03:32:06.258381Z","iopub.status.idle":"2024-05-28T03:32:06.269968Z","shell.execute_reply.started":"2024-05-28T03:32:06.258348Z","shell.execute_reply":"2024-05-28T03:32:06.268608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score\": data_submission['score']\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-28T03:33:23.190527Z","iopub.execute_input":"2024-05-28T03:33:23.190951Z","iopub.status.idle":"2024-05-28T03:33:23.201588Z","shell.execute_reply.started":"2024-05-28T03:33:23.190919Z","shell.execute_reply":"2024-05-28T03:33:23.200201Z"},"trusted":true},"execution_count":null,"outputs":[]}]}