{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import Python modules + Load Data (we decided to do it in the kaggle kernel)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-19T06:27:09.283525Z","iopub.execute_input":"2024-04-19T06:27:09.284218Z","iopub.status.idle":"2024-04-19T06:27:10.918486Z","shell.execute_reply.started":"2024-04-19T06:27:09.284140Z","shell.execute_reply":"2024-04-19T06:27:10.917247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> General\n# warnings - to manage irrelevant warnings when running the code\nimport warnings\n# Ignore (that is, do not print) warnings\nwarnings.filterwarnings(\"ignore\")\n# matplotlib's pyplot - for general plotting\nimport matplotlib.pyplot as plt\n\n# ---> Manage data\n# NumPy - for basic mathematical operations\nimport numpy as np\n# polars - to efficiently manage large quantities of data and to deal with tables\nimport polars as pl\n# Pandas - to deal with tables \nimport pandas as pd\n\n# ---> Modelling\n# LightGBM - for applying gradient-boosting algorithms\nimport lightgbm as lgb\n# Import some scikit-learn's metrics \nfrom sklearn.metrics import (\n    accuracy_score,\n    roc_curve,\n    roc_auc_score\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:10.920831Z","iopub.execute_input":"2024-04-19T06:27:10.922505Z","iopub.status.idle":"2024-04-19T06:27:15.047950Z","shell.execute_reply.started":"2024-04-19T06:27:10.922455Z","shell.execute_reply":"2024-04-19T06:27:15.046430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path to data directory in the kaggle kernel\npath_data = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:15.049565Z","iopub.execute_input":"2024-04-19T06:27:15.050246Z","iopub.status.idle":"2024-04-19T06:27:15.057396Z","shell.execute_reply.started":"2024-04-19T06:27:15.050205Z","shell.execute_reply":"2024-04-19T06:27:15.056010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create polars dataframes from selected training and test data","metadata":{}},{"cell_type":"code","source":"# ---> Function for setting polars dataframes dtypes\ndef set_pl_dtypes(df):\n    for col in df.columns:\n        # If column is associated with P or A-type transforms, set its dtype to to\n        # Float64\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64))\n        # If column is associated with M-type transform, set its dtype to to Categorical\n        if col[-1] in (\"M\"):\n            df = df.with_columns(pl.col(col).cast(pl.Categorical))\n        # If column is associated with D-type transform, set its dtype to to Date\n        if col[-1] in (\"D\"):\n            df = df.with_columns(pl.col(col).cast(pl.Date))\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:15.060636Z","iopub.execute_input":"2024-04-19T06:27:15.062042Z","iopub.status.idle":"2024-04-19T06:27:15.076127Z","shell.execute_reply.started":"2024-04-19T06:27:15.061988Z","shell.execute_reply":"2024-04-19T06:27:15.074836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Create polars dataframes for training data\n\n# Define polars dataframe containing data from base training CSV file\ndf_train_base = pl.read_csv(path_data + \"csv_files/train/train_base.csv\")\n\n# Define polars dataframe containing data from static internal training CSV files 0 and\n# 1\ndf_train_static = pl.concat([pl.read_csv(path_data + \"csv_files/train/train_static_0_0.csv\").pipe(set_pl_dtypes),\n                             pl.read_csv(path_data + \"csv_files/train/train_static_0_1.csv\").pipe(set_pl_dtypes)],\n                            # vertically concatenate, while conveniently redefining\n                            # column's data types to support the data from common\n                            # columns\n                            how=\"vertical_relaxed\")\n\n# Define polars dataframe containing data from static external (from a Credit Bureau)\n# training CSV files\ndf_train_static_cb = pl.read_csv(path_data + \"csv_files/train/train_static_cb_0.csv\").pipe(set_pl_dtypes)\n\n# Define polars dataframe containing data from the persons' training CSV file of depth 1\ndf_train_person_1 = pl.read_csv(path_data + \"csv_files/train/train_person_1.csv\").pipe(set_pl_dtypes)\n\n# Define polars dataframe containing data from Credit Bureau B's training CSV file of\n# depth 2\ndf_train_credit_bureau_b_2 = pl.read_csv(path_data + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_pl_dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:15.077606Z","iopub.execute_input":"2024-04-19T06:27:15.077984Z","iopub.status.idle":"2024-04-19T06:27:44.012228Z","shell.execute_reply.started":"2024-04-19T06:27:15.077955Z","shell.execute_reply":"2024-04-19T06:27:44.010422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Create polars dataframes for test data\n\n# Define polars dataframe containing data from base test CSV file\ndf_test_base = pl.read_csv(path_data + \"csv_files/test/test_base.csv\")\n\n# Define polars dataframe containing data from static internal test CSV files 0 and\n# 1\ndf_test_static = pl.concat(\n    [pl.read_csv(path_data + \"csv_files/test/test_static_0_0.csv\").pipe(set_pl_dtypes),\n     pl.read_csv(path_data + \"csv_files/test/test_static_0_1.csv\").pipe(set_pl_dtypes),\n     pl.read_csv(path_data + \"csv_files/test/test_static_0_2.csv\").pipe(set_pl_dtypes)],\n    # vertically concatenate, while conveniently redefining column's data types to\n    # support the data from common columns\n    how=\"vertical_relaxed\")\n# Define polars dataframe containing data from static external (from a Credit Bureau)\n# test CSV files\ndf_test_static_cb = pl.read_csv(path_data + \"csv_files/test/test_static_cb_0.csv\").pipe(set_pl_dtypes)\n\n# Define polars dataframe containing data from the persons' test CSV file of depth 1\ndf_test_person_1 = pl.read_csv(path_data + \"csv_files/test/test_person_1.csv\").pipe(set_pl_dtypes)\n\n# Define polars dataframe containing data from Credit Bureau B's test CSV file of depth\n# 2\ndf_test_credit_bureau_b_2 = pl.read_csv(path_data + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_pl_dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:44.014383Z","iopub.execute_input":"2024-04-19T06:27:44.017784Z","iopub.status.idle":"2024-04-19T06:27:44.101129Z","shell.execute_reply.started":"2024-04-19T06:27:44.017732Z","shell.execute_reply":"2024-04-19T06:27:44.099603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Define selecting columns for the static dataframes\n\n# List of labels of the selected columns of training and test static internal dataframes\n# [NOTE: for the sake of simplicity, solely A and M-types are herein selected. These\n# stand for \"transform amount\" and \"masking categories\" of the groups of transforms,\n# with capital letters \"A\" and \"M\" as the last characters of the labels, respectively.]\ncols_static_selec = []\nfor col in df_train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        cols_static_selec.append(col)\n\n# List of labels of the selected columns of training and test static Credit Bureau\n# dataframes\n# [NOTE: for the sake of simplicity, solely A and M-types are herein selected.]\ncols_static_cb_selec = []\nfor col in df_train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        cols_static_cb_selec.append(col)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:44.102919Z","iopub.execute_input":"2024-04-19T06:27:44.103477Z","iopub.status.idle":"2024-04-19T06:27:44.112420Z","shell.execute_reply.started":"2024-04-19T06:27:44.103432Z","shell.execute_reply":"2024-04-19T06:27:44.110781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Filter training dataframes\n\n# Group df_train_person_1 by case_id (that is, collapse rows with same case_id value,\n# which with an additional \"aggregation\" (to be performed later) makes the entries of\n# the chosen columns to accomodate lists with the respective collapsed values (or\n# operations on them, as wished))\ndf_train_person_1_group = df_train_person_1.group_by(\"case_id\")\n\n\n# Aggregate to the group, the maximum income (from main occupation) of the set of people\n# associated with each group, as well as a boolean that holds true if any of the set of\n# people of each group is self-employed\ndf_train_person_1_1 = df_train_person_1_group.agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_max_A\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").any().alias(\"anyselfemployed_T\")\n)\n\n# Select from client df_train_person_1 columns \"case_id\", \"num_group1\" and\n# \"housetype_905L\". Keep solely rows associated with applicants (num_group1 = 0), drop\n# the column \"num_group1\" and rename \"housetype_905L\" to \"housetype_applicant_L\", since\n# it refers now to the house type (e.g. \"owned\") of the applicants.\ndf_train_person_1_2 = df_train_person_1.select(\n    [\"case_id\", \"num_group1\", \"housetype_905L\"]\n).filter(pl.col(\"num_group1\") == 0).drop(\"num_group1\").rename(\n    {\"housetype_905L\": \"housetype_applicant_L\"}\n)\n\n# Group df_train_credit_bureau_b_2 by case_id and aggregate maximum value of the number\n# of overdue payments (pmts_pmtsoverdue_635A) and a boolean that holds true if there is\n# any number of days of overdue payment greater than 31.\ndf_train_credit_bureau_b_2 = df_train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_max_A\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).any().alias(\"pmts_dpdvalue_anyover31_P\")\n)\n\n# Training dataframe defined by a left join of all training dataframes.\ndf_train = df_train_base\\\n.join(df_train_static.select([\"case_id\"] + cols_static_selec),\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_train_static_cb.select([\"case_id\"] + cols_static_cb_selec),\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_train_person_1_1,\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_train_person_1_2,\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_train_credit_bureau_b_2,\n      how=\"left\",\n      on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:44.114304Z","iopub.execute_input":"2024-04-19T06:27:44.114910Z","iopub.status.idle":"2024-04-19T06:27:49.423205Z","shell.execute_reply.started":"2024-04-19T06:27:44.114861Z","shell.execute_reply":"2024-04-19T06:27:49.422173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Filter test dataframes\n\n# Group df_test_person_1 by case_id (that is, collapse rows with same case_id value,\n# which with an additional \"aggregation\" (to be performed later) makes the entries of\n# the chosen columns to accomodate lists with the respective collapsed values (or\n# operations on them, as wished))\ndf_test_person_1_group = df_test_person_1.group_by(\"case_id\")\n\n\n# Aggregate to the group, the maximum income (from main occupation) of the set of people\n# associated with each group, as well as a boolean that holds true if any of the set of\n# people of each group is self-employed\ndf_test_person_1_1 = df_test_person_1_group.agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_max_A\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").any().alias(\"anyselfemployed_T\")\n)\n\n# Select from client df_test_person_1 columns \"case_id\", \"num_group1\" and\n# \"housetype_905L\". Keep solely rows associated with applicants (num_group1 = 0), drop\n# the column \"num_group1\" and rename \"housetype_905L\" to \"housetype_applicant_L\", since\n# it refers now to the house type (e.g. \"owned\") of the applicants.\ndf_test_person_1_2 = df_test_person_1.select(\n    [\"case_id\", \"num_group1\", \"housetype_905L\"]\n).filter(pl.col(\"num_group1\") == 0).drop(\"num_group1\").rename(\n    {\"housetype_905L\": \"housetype_applicant_L\"}\n)\n\n# Group df_test_credit_bureau_b_2 by case_id and aggregate maximum value of the number\n# of overdue payments (pmts_pmtsoverdue_635A) and a boolean that holds true if there is\n# any number of days of overdue payment greater than 31.\ndf_test_credit_bureau_b_2 = df_test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_max_A\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).any().alias(\"pmts_dpdvalue_anyover31_P\")\n)\n\n# Submission dataframe defined by a left join of all test dataframes.\n# [NOTE: the label \"submission\" is used in place of \"test\" as another test dataset is to\n# be created from the training one.]\ndf_submission = df_test_base\\\n.join(df_test_static.select([\"case_id\"] + cols_static_selec),\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_test_static_cb.select([\"case_id\"] + cols_static_cb_selec),\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_test_person_1_1,\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_test_person_1_2,\n      how=\"left\",\n      on=\"case_id\")\\\n.join(df_test_credit_bureau_b_2,\n      how=\"left\",\n      on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.424241Z","iopub.execute_input":"2024-04-19T06:27:49.424559Z","iopub.status.idle":"2024-04-19T06:27:49.444384Z","shell.execute_reply.started":"2024-04-19T06:27:49.424532Z","shell.execute_reply":"2024-04-19T06:27:49.443154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display first five entries of the training dataframe\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.449646Z","iopub.execute_input":"2024-04-19T06:27:49.450201Z","iopub.status.idle":"2024-04-19T06:27:49.483485Z","shell.execute_reply.started":"2024-04-19T06:27:49.450152Z","shell.execute_reply":"2024-04-19T06:27:49.482242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define training, validation and test datasets from the original training one","metadata":{}},{"cell_type":"code","source":"# ---> Info on the split of the original training dataset\n\n# Dictionary of parameters for splitting the original training dataset into a new\n# training, validation and test datasets\ntrain_valid_test_split = {\n    # Fraction of the original training dataset to be used on validation\n    \"valid_frac\": 0.2,\n    # Fraction of the original training dataset to be used on testing\n    \"test_frac\": 0.2\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.485040Z","iopub.execute_input":"2024-04-19T06:27:49.485436Z","iopub.status.idle":"2024-04-19T06:27:49.492830Z","shell.execute_reply.started":"2024-04-19T06:27:49.485404Z","shell.execute_reply":"2024-04-19T06:27:49.490759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"case_id_train_old = df_train[\"case_id\"].unique().shuffle(seed=42)\n\n# Number of rows in the original training dataset\nN_train_old = len(case_id_train_old)\n\n# Number of rows in the new original training dataset\nN_train = int((1 - train_valid_test_split[\"valid_frac\"] -\n               train_valid_test_split[\"test_frac\"]) * N_train_old)\n\n# Number of rows in the new validation dataset\nN_valid = int(train_valid_test_split[\"valid_frac\"] * N_train_old)\n\n# polars series of case_id in the new training dataframe\ncase_id_train = case_id_train_old.head(N_train)\n\n# polars series of case_id in the new validation dataframe\ncase_id_valid = case_id_train_old.tail(-N_train).head(N_valid)\n\n# polars series of case_id in the new test dataframe\ncase_id_test = case_id_train_old.tail(-N_train).tail(-N_valid)\n\n# polars series of case_id in the submission dataframe\ncase_id_submission = df_submission[\"case_id\"].unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.494670Z","iopub.execute_input":"2024-04-19T06:27:49.495042Z","iopub.status.idle":"2024-04-19T06:27:49.640235Z","shell.execute_reply.started":"2024-04-19T06:27:49.495013Z","shell.execute_reply":"2024-04-19T06:27:49.638693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of labels of the columns of the dataframes which are associated with features\n# [NOTE: features are such that the labels of the respective columns have lower case\n# characters in their whole extent except in the last position.]\ncols_x = []\nfor col in df_train.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_x.append(col)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.642042Z","iopub.execute_input":"2024-04-19T06:27:49.642436Z","iopub.status.idle":"2024-04-19T06:27:49.650311Z","shell.execute_reply.started":"2024-04-19T06:27:49.642406Z","shell.execute_reply":"2024-04-19T06:27:49.648698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Define auxiliary functions\n\n# Function that returns a dictionary of pandas dataframes from given polars dataframe\n# and list of case indices\n# [NOTE: lightgbm (the package used for training the model) only supports pandas\n# dataframes, hence the need to convert the polars dataframes to pandas'.]\ndef from_polars_to_pandas(df, case_id, cols_x):\n    # polars dataframe corresponding to df for the given case_id\n    df_case_id = df.filter(pl.col(\"case_id\").is_in(case_id))\n    \n    # Dictionary of pandas dataframes\n    dt = {\n        # Base dataframe with case_id and columns that are required for the computation\n        # of the Gini coefficient\n        \"base\": df_case_id[[\"case_id\", \"WEEK_NUM\"]].to_pandas(),\n        # Features' dataframe\n        \"x\": df_case_id[cols_x].to_pandas()\n    }\n    \n    # If the polars dataframe has labels (column \"target\") add them to the dictioanry\n    if \"target\" in df_case_id.columns:\n        dt[\"base\"].insert(loc=dt[\"base\"].shape[1], column=\"y\", value=df_case_id[\"target\"])\n        dt[\"y\"] = df_case_id[\"target\"].to_pandas()\n\n    return dt\n\n# Function for converting object columns of pandas dataframes to category columns\n# [NOTE: if there is a column of the tuple of dataframes that is type \"object\", the\n# respective columns in the that and the other dataframes are converted to the type\n# category.]\n# [NOTE: a column of type \"object\" is such that it has entries of solely type \"str\" or\n# entries of multiple different types (e.g. \"NoneType\" and \"float\").]\n# [NOTE: category columns are columns whose entries pertain to a finite list of text\n# values.]\n# [NOTE: the category \"Unknown\" is added to the list of categories to support the case\n# in which validation and test dataframes have exclusive categories not pertaining to\n# the training dataframe - these exclusive categories which are not supported by the\n# trained model should be replaced by the \"Unknown\" category.]\ndef convert_cols_obj_to_cols_cat(*dfs):\n    # List of columns of the tuple of dataframes that are of type \"object\" in at least\n    # one of the dataframes\n    cols_object = list(set().union(*(df.select_dtypes(include=[\"object\"]).columns for df in dfs)))\n    # For each column of dtype \"object\"\n    for col in cols_object:\n        # For each dataframe of the tuple\n        for df in dfs:\n            # Convert current column to dtype \"category\"\n            df[col] = df[col].astype(\"category\")\n            # New categorical dtype whose categories correspond to the ones of the\n            # current column and the category \"Unknown\", being ordered\n            new_dtype = pd.CategoricalDtype(categories=df[col].cat.categories.to_list() +\n                                            [\"Unknown\"],\n                                            ordered=True)\n            # Assign new dtype to current column\n            df[col] = df[col].astype(new_dtype)\n    return dfs\n\n# Function for making categories of some pandas dataframe that do not pertain to a\n# reference one be replaced by the category \"Unknown\"\ndef make_cat_excl_unknown(df, df_ref):\n    # For each categorical column of the reference pandas dataframe\n    for col in df_ref.select_dtypes(include=[\"category\"]).columns:\n        # List of categories in the reference pandas dataframe\n        cat_ref = df_ref[col].cat.categories.to_list()\n        # List of categories in the pandas dataframe of interest\n        cat = df[col].cat.categories.to_list()\n        # List of common categories\n        cat_common = list(set(cat).intersection(cat_ref))\n        # List of exclusive categories\n        cat_exc = list(set(cat).difference(cat_common))\n        # New categorical dtype whose categories correspond to the common ones\n        new_dtype = pd.CategoricalDtype(categories=cat_common,\n                                        ordered=True)\n        # Replace current column's entries associated with exclusive categories as\n        # \"Unknown\"\n        df[col] = df[col].replace(to_replace=cat_exc, value=\"Unknown\")\n        # Assign the new dtype to the current column\n        df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.652456Z","iopub.execute_input":"2024-04-19T06:27:49.653077Z","iopub.status.idle":"2024-04-19T06:27:49.672351Z","shell.execute_reply.started":"2024-04-19T06:27:49.653026Z","shell.execute_reply":"2024-04-19T06:27:49.670905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Define dictionaries of pandas dataframes associated with the training,\n# validation, test and submission data\n\n# Dictionary of pandas dataframes associated with the training data\ndt_train = from_polars_to_pandas(df_train, case_id_train, cols_x)\n# Dictionary of pandas dataframes associated with the validation data\ndt_valid = from_polars_to_pandas(df_train, case_id_valid, cols_x)\n# Dictionary of pandas dataframes associated with the test data\ndt_test = from_polars_to_pandas(df_train, case_id_test, cols_x)\n# Dictionary of pandas dataframes associated with the submission data\ndt_submission = from_polars_to_pandas(df_submission, case_id_submission, cols_x)\n\n# Convert object columns of feature pandas dataframes to category columns and also add\n# the category \"Unknown\"\n# [NOTE: if there is a column of the tuple of dataframes that is type \"object\", the\n# respective columns in the that and the other dataframes are converted to the type\n# category.]\n(dt_train[\"x\"], dt_valid[\"x\"], dt_test[\"x\"], dt_submission[\"x\"]) = convert_cols_obj_to_cols_cat(\n    dt_train[\"x\"], dt_valid[\"x\"], dt_test[\"x\"], dt_submission[\"x\"]\n)\n# For compatibility reasons, make categories of the feature validation, test and\n# submission dataframes which do not pertain to the the training dataframe be replaced\n# by the category \"Unknown\"\ndt_valid[\"x\"] = make_cat_excl_unknown(df=dt_valid[\"x\"], df_ref=dt_train[\"x\"])\ndt_test[\"x\"] = make_cat_excl_unknown(df=dt_test[\"x\"], df_ref=dt_train[\"x\"])\ndt_submission[\"x\"] = make_cat_excl_unknown(df=dt_submission[\"x\"], df_ref=dt_train[\"x\"])\n\n# Display shapes of feature pandas dataframes\ndisplay(pd.DataFrame(data=\n                     {\"Feature dataset\": [\"train\", \"valid\", \"test\", \"submission\"],\n                      \"N_rows\": [dt[\"x\"].shape[0]for\n                                          dt in (dt_train, dt_valid, dt_test, dt_submission)],\n                      \"N_cols\": [dt[\"x\"].shape[1]for\n                                          dt in (dt_train, dt_valid, dt_test, dt_submission)]\n                     }))","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:49.674147Z","iopub.execute_input":"2024-04-19T06:27:49.674550Z","iopub.status.idle":"2024-04-19T06:27:51.396735Z","shell.execute_reply.started":"2024-04-19T06:27:49.674518Z","shell.execute_reply":"2024-04-19T06:27:51.395313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"En resumen, tu conjunto de datos de entrenamiento (dt_train) es el que usarás para realizar la preparación de datos y entrenar tu modelo. Los conjuntos de datos de validación (dt_valid) y prueba (dt_test) se utilizarán para evaluar el rendimiento del modelo, mientras que el conjunto de datos de presentación (dt_submission) se utilizará para generar las predicciones finales que enviarás.","metadata":{}},{"cell_type":"markdown","source":"# PREPROCESSING","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:27:51.398636Z","iopub.execute_input":"2024-04-19T06:27:51.399168Z","iopub.status.idle":"2024-04-19T06:27:51.419010Z","shell.execute_reply.started":"2024-04-19T06:27:51.399114Z","shell.execute_reply":"2024-04-19T06:27:51.416798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display first five entries of the test dataframe\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:36:23.077744Z","iopub.execute_input":"2024-04-19T06:36:23.078279Z","iopub.status.idle":"2024-04-19T06:36:23.097093Z","shell.execute_reply.started":"2024-04-19T06:36:23.078247Z","shell.execute_reply":"2024-04-19T06:36:23.095467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Info on the split of the original training dataset\n\n# Dictionary of parameters for splitting the original training dataset into a new\n# training, validation and test datasets\ntrain_valid_test_split = {\n    # Fraction of the original training dataset to be used on validation\n    \"valid_frac\": 0.2,\n    # Fraction of the original training dataset to be used on testing\n    \"test_frac\": 0.2\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:36:44.380473Z","iopub.execute_input":"2024-04-19T06:36:44.380952Z","iopub.status.idle":"2024-04-19T06:36:44.387793Z","shell.execute_reply.started":"2024-04-19T06:36:44.380897Z","shell.execute_reply":"2024-04-19T06:36:44.386308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ---> Define polars series of case_id for the training, validation, test and submission\n# dataframes\n\n# polars series of unique values of case_id in the original training dataframe, suffle\n# with seed 42\n# [NOTE: unique values are taken in order to disregard possible duplicates.]\n# [NOTE: shuffle is done to randomly distribute the training, validation and test\n# datasets at an upcoming moment.]\n# [NOTE: by issuing a seed number to the shuffle random process, later calls of such\n# process would produce the same result, ensuring reproducibility of this work.]\ncase_id_train_old = df_train[\"case_id\"].unique().shuffle(seed=42)\n\n# Number of rows in the original training dataset\nN_train_old = len(case_id_train_old)\n\n# Number of rows in the new original training dataset\nN_train = int((1 - train_valid_test_split[\"valid_frac\"] -\n               train_valid_test_split[\"test_frac\"]) * N_train_old)\n\n# Number of rows in the new validation dataset\nN_valid = int(train_valid_test_split[\"valid_frac\"] * N_train_old)\n\n# polars series of case_id in the new training dataframe\ncase_id_train = case_id_train_old.head(N_train)\n\n# polars series of case_id in the new validation dataframe\ncase_id_valid = case_id_train_old.tail(-N_train).head(N_valid)\n\n# polars series of case_id in the new test dataframe\ncase_id_test = case_id_train_old.tail(-N_train).tail(-N_valid)\n\n# polars series of case_id in the submission dataframe\ncase_id_submission = df_submission[\"case_id\"].unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:37:01.562568Z","iopub.execute_input":"2024-04-19T06:37:01.563045Z","iopub.status.idle":"2024-04-19T06:37:01.697065Z","shell.execute_reply.started":"2024-04-19T06:37:01.563013Z","shell.execute_reply":"2024-04-19T06:37:01.695742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of labels of the columns of the dataframes which are associated with features\n# [NOTE: features are such that the labels of the respective columns have lower case\n# characters in their whole extent except in the last position.]\ncols_x = []\nfor col in df_train.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_x.append(col)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:37:11.503628Z","iopub.execute_input":"2024-04-19T06:37:11.504397Z","iopub.status.idle":"2024-04-19T06:37:11.512763Z","shell.execute_reply.started":"2024-04-19T06:37:11.504359Z","shell.execute_reply":"2024-04-19T06:37:11.511375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LightGBM's training dataset\nds_train = lgb.Dataset(\n    data=dt_train[\"x\"],\n    label=dt_train[\"y\"]\n)\n\n# LightGBM's validation dataset\n# [NOTE: by setting reference=ds_train, the defined dataset is regarded as a validation\n# one for ds_train.]\nds_valid = lgb.Dataset(\n    data=dt_valid[\"x\"],\n    label=dt_valid[\"y\"],\n    reference=ds_train\n)\n\n# Dictionary of parameters for training the LightGBM model\nparams = {\n    # Gradient Boosting type\n    # [NOTE: possible values:\n    #  * \"gbdt\", standing for \"Gradient Boosting Decision Tree\",\n    #  * \"dart\", standing for \"Dropouts meet multiple Additive Regression Trees\",\n    #  * \"rf\", standing for \"Random Forest\",\n    # .],\n    \"boosting_type\": \"gbdt\",\n    # Learning objective\n    # [NOTE: \"binary\" stands for binary classification.]\n    \"objective\": \"binary\",\n    # List of metrics to be evaluated while training\n    # [NOTE: \"auc\" stands for \"Area Under the Curve\" - this curve is the \"Receiver\n    # Operating Characterisitc\" (ROC).]\n    # [NOTE: \"cross_entropy\" is the cross-entropy cost function.]\n    \"metric\": [\"auc\", \"cross_entropy\"],\n    # Max depth of the weak tree learners\n    \"max_depth\": 3,\n    # Maximum number of leave nodes in a weak tree learner\n    \"num_leaves\": 31,\n    # Learning rate\n    \"learning_rate\": 0.05,\n    # Fraction of feature components to use in each Gradient Boosting iteration\n    # [NOTE: LightGBM will randomly select a subset of feature components on each\n    # iteration (tree). The dimensionality of this subset is a fraction feature_fraction\n    # of the dimensionality of the whole feature vector.]\n    # [NOTE: feature_fraction must pertain to the interval ]0, 1].]\n    # [NOTE: the default value is 1.]\n    # [NOTE: feature_fraction may be used to speed up training and to avoid getting an\n    # overfitting model.]\n    \"feature_fraction\": 0.9,\n    # Fraction of data to use in each Gradient Boosting iteration\n    # [NOTE: LightGBM will randomly select a subset of training points on each\n    # iteration (tree). The cardinality of this subset is a fraction bagging_fraction\n    # of the cardinality of the whole training set.]\n    # [NOTE: to consider bagging, the bagging frequency parameter (bagging_freq) must be\n    # set to a non-null value (by default it is null).]\n    # [NOTE: bagging_fraction must pertain to the interval ]0, 1].]\n    # [NOTE: the default value is 1.]\n    # [NOTE: bagging_fraction may be used to speed up training and to avoid getting an\n    # overfitting model.]\n    \"bagging_fraction\": 0.8,\n    # Bagging frequency\n    # [NOTE: A bagging frequency of k means that bagging is done at each k iterations\n    # (trees) and that resultant subset of training points is used in the current\n    # iteration and in the next k-1.]\n    # [NOTE: the default value is 0.]\n    \"bagging_freq\": 5,\n    # Maximum number of weak learners (the same as num_iterations)\n    \"n_estimators\": 1000,\n    # Verbosity level of LightGBM\n    # [NOTE: \"-1\" means solely \"fatal errors\".]\n    \"verbose\": -1\n}\n\n\n# Dictionary of evaluation results of the model (to be used later for plotting)\nmodel_eval_result = {}\n\n# Trained Booster model\nmodel = lgb.train(\n    params=params,\n    train_set=ds_train,\n    # List of LightGBM datasets whose points are used to evaluate the model while\n    # training\n    # [NOTE: since ds_train was already used in the argument of train_set, lgb.train\n    # will recognise it as the training dataset and early stopping would only be applied\n    # on the validation one, ds_valid. Anyway, if the user wants to get lgb.train to\n    # compute the metrics for the training dataset, it needs to be included in the\n    # valid_sets argument.]\n    valid_sets=[ds_valid, ds_train],\n    # List of names associated with the list valid_sets\n    valid_names=[\"valid\", \"train\"],\n    # List of callback functions that are applied at each iteration\n    callbacks=[\n        # Report evaluation results at each 50 iterations\n        lgb.log_evaluation(period=50),\n        # Enable early stopping (the model will train until the validation score doesn’t\n        # improve by at least min_delta in the last stopping_rounds-1 iterations and in\n        # the current one)\n        lgb.early_stopping(\n            # Boolean that if set to True makes the trainer to solely use the first\n            # metric in the list of metrics for performing early stopping\n            first_metric_only=True,\n            stopping_rounds=10,\n            verbose=True,\n            min_delta=0),\n        # Save evaluation results into a dictionary\n        lgb.record_evaluation(eval_result=model_eval_result)\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-19T06:42:43.222148Z","iopub.execute_input":"2024-04-19T06:42:43.222675Z","iopub.status.idle":"2024-04-19T06:44:25.099075Z","shell.execute_reply.started":"2024-04-19T06:42:43.222643Z","shell.execute_reply":"2024-04-19T06:44:25.097579Z"},"trusted":true},"execution_count":null,"outputs":[]}]}