{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nfrom glob import glob\nimport gc\n\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:10:13.910429Z","iopub.execute_input":"2024-04-22T15:10:13.911172Z","iopub.status.idle":"2024-04-22T15:10:18.712384Z","shell.execute_reply.started":"2024-04-22T15:10:13.911132Z","shell.execute_reply":"2024-04-22T15:10:18.711231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:10:18.714258Z","iopub.execute_input":"2024-04-22T15:10:18.715463Z","iopub.status.idle":"2024-04-22T15:10:18.720050Z","shell.execute_reply.started":"2024-04-22T15:10:18.715429Z","shell.execute_reply":"2024-04-22T15:10:18.719194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Basic flow of Kaggle competitions","metadata":{}},{"cell_type":"markdown","source":"# 1. Understand the problem and the data ","metadata":{}},{"cell_type":"markdown","source":"## Readings","metadata":{}},{"cell_type":"markdown","source":"- https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability\n- https://www.geeksforgeeks.org/auc-roc-curve/\n(AUC is a probability that any pair of samples will be correctly ordered in terms of default probability)","metadata":{}},{"cell_type":"markdown","source":"Praca domowa 2024-03-11: przeczytac\n- https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/476449\n- https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/482474","metadata":{}},{"cell_type":"markdown","source":"# 2. First (sample) submission","metadata":{}},{"cell_type":"code","source":"sample_sub = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:10:18.722569Z","iopub.execute_input":"2024-04-22T15:10:18.723201Z","iopub.status.idle":"2024-04-22T15:10:18.770877Z","shell.execute_reply.started":"2024-04-22T15:10:18.723170Z","shell.execute_reply":"2024-04-22T15:10:18.769801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:10:18.773801Z","iopub.execute_input":"2024-04-22T15:10:18.774547Z","iopub.status.idle":"2024-04-22T15:10:18.797776Z","shell.execute_reply.started":"2024-04-22T15:10:18.774512Z","shell.execute_reply":"2024-04-22T15:10:18.796562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### / First (sample) submission","metadata":{}},{"cell_type":"markdown","source":"# 3. Establish validation framework and train the first model","metadata":{}},{"cell_type":"markdown","source":"## Read the data","metadata":{}},{"cell_type":"code","source":"%%time\ntr_base = pl.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_base.parquet')\ntest_base = pl.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test/test_base.parquet')\n\n# train_static_00 = pl.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_0_0.parquet')\n# train_static_01 = pl.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_0_1.parquet')\n# train_static = pl.concat([train_static_00, train_static_01], how = 'vertical_relaxed')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:10:18.799358Z","iopub.execute_input":"2024-04-22T15:10:18.800547Z","iopub.status.idle":"2024-04-22T15:10:19.171934Z","shell.execute_reply.started":"2024-04-22T15:10:18.800509Z","shell.execute_reply":"2024-04-22T15:10:19.170385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_static(dataset='train'):\n    # Define the pattern to match all relevant parquet files\n    pattern = f'/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/{dataset}/{dataset}_static_0_*.parquet'\n    \n    # Use glob to find all files matching the pattern\n    files = glob(pattern)\n    \n    # Load and concatenate all matching files\n    dfs = [pl.read_parquet(file) for file in files]\n    combined_df = pl.concat(dfs, how='vertical_relaxed')\n    \n    return combined_df","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:10:19.173711Z","iopub.execute_input":"2024-04-22T15:10:19.174294Z","iopub.status.idle":"2024-04-22T15:10:19.186888Z","shell.execute_reply.started":"2024-04-22T15:10:19.174245Z","shell.execute_reply":"2024-04-22T15:10:19.185153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static = read_static('train')\ntest_static = read_static('test')","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:31.556046Z","iopub.execute_input":"2024-04-22T15:13:31.556507Z","iopub.status.idle":"2024-04-22T15:13:35.971608Z","shell.execute_reply.started":"2024-04-22T15:13:31.556474Z","shell.execute_reply":"2024-04-22T15:13:35.970279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_base_merged = tr_base.join(train_static, how='left', on = 'case_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:35.974444Z","iopub.execute_input":"2024-04-22T15:13:35.974980Z","iopub.status.idle":"2024-04-22T15:13:37.399526Z","shell.execute_reply.started":"2024-04-22T15:13:35.974937Z","shell.execute_reply":"2024-04-22T15:13:37.398602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_base_merged = test_base.join(test_static, how='left', on = 'case_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.400463Z","iopub.execute_input":"2024-04-22T15:13:37.400772Z","iopub.status.idle":"2024-04-22T15:13:37.406484Z","shell.execute_reply.started":"2024-04-22T15:13:37.400746Z","shell.execute_reply":"2024-04-22T15:13:37.405701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Praca Domowa 2024-03-18:\n    - Przeczytac feature definitions dla kolumn z tabeli train_static_00","metadata":{}},{"cell_type":"code","source":"feature_def = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv')\nfeature_def[feature_def['Variable']=='annuity_780A']['Description']","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.409289Z","iopub.execute_input":"2024-04-22T15:13:37.410268Z","iopub.status.idle":"2024-04-22T15:13:37.441991Z","shell.execute_reply.started":"2024-04-22T15:13:37.410238Z","shell.execute_reply":"2024-04-22T15:13:37.440360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Praca domowa 2024-03-25:\n    - Wytrenowac model LGBM, ktory przyjmuje y: 'target', X: kolumny na prawo od 'target'\n    - Na tabeli tr_base_merged","metadata":{}},{"cell_type":"markdown","source":"### Train the model (Kudos to Milosz Goszczynski)","metadata":{}},{"cell_type":"code","source":"cols_types = list(zip(tr_base_merged.dtypes, tr_base_merged.columns))\nnonstring_cols = [c[1] for c in cols_types if c[0] != pl.String][4:]\nstring_cols = [c[1] for c in cols_types if c[0] == pl.String]\nstring_cols = [c for c in string_cols if (c[-1]!= 'D')&('date' not in c)]","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.443855Z","iopub.execute_input":"2024-04-22T15:13:37.444184Z","iopub.status.idle":"2024-04-22T15:13:37.451321Z","shell.execute_reply.started":"2024-04-22T15:13:37.444156Z","shell.execute_reply":"2024-04-22T15:13:37.450181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"string_cols","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.452542Z","iopub.execute_input":"2024-04-22T15:13:37.453475Z","iopub.status.idle":"2024-04-22T15:13:37.469521Z","shell.execute_reply.started":"2024-04-22T15:13:37.453446Z","shell.execute_reply":"2024-04-22T15:13:37.468085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical variables","metadata":{}},{"cell_type":"markdown","source":"### Simple int encoding","metadata":{}},{"cell_type":"markdown","source":"### Simple int encoding","metadata":{}},{"cell_type":"code","source":"# from sklearn.preprocessing import OrdinalEncoder\n# le = OrdinalEncoder()\n# encoded = le.fit_transform(tr_base_merged.select(string_cols)).astype(np.int16)\n\n# for i, col_name in enumerate(string_cols):\n#     # Replace each column in X with its encoded counterpart\n#     tr_base_merged = tr_base_merged.with_columns(pl.Series(name=col_name, values=encoded[:, i]))","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.470927Z","iopub.execute_input":"2024-04-22T15:13:37.471790Z","iopub.status.idle":"2024-04-22T15:13:37.476605Z","shell.execute_reply.started":"2024-04-22T15:13:37.471760Z","shell.execute_reply":"2024-04-22T15:13:37.475713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Target encoding","metadata":{}},{"cell_type":"code","source":"class CategoricalEncoder():\n    def __init__(self, na_value = -1):\n        self.full_cat_encoding_dicts = {}\n        self.na_value = na_value\n        self.cat_features = []\n        self.target_col = ''\n    \n    def transform_test(self, df):\n        for cur_column in self.cat_features: \n            replace_dict = self.full_cat_encoding_dicts[cur_column]\n\n            mask = ~pl.col(cur_column).is_in(replace_dict.keys())\n\n            df = df.with_columns(\n                pl.when(mask)\n                  .then(self.na_value)\n                  .otherwise(pl.col(cur_column))\n                  .alias(cur_column)\n            )\n\n            df = df.with_columns(\n                pl.col(cur_column)\n                  .replace(replace_dict)\n                  .cast(pl.Float32)\n                  .alias(cur_column))\n        return df\n\n    def fit_transform_train(self, df, cat_features, target_col, n_folds=5):\n        \n        self.cat_features = cat_features\n        self.target_col = target_col\n        \n        #fitting on full train for test\n        for cur_column in self.cat_features:\n            cur_grouped_table = df.group_by(cur_column).agg(pl.mean(target_col))\n            replace_dict = {row[0] : row[1] for row in cur_grouped_table.iter_rows()}\n            self.full_cat_encoding_dicts[cur_column] = replace_dict\n        \n        #fitting on cross-val train for training\n        kf_splitter = KFold(n_splits = n_folds, shuffle = True, random_state = 101)\n\n        indices = []\n        for train_index, valid_index in kf_splitter.split(df):\n            indices.append((train_index, valid_index))\n\n        full_encoded_table = []\n        original_indices = []\n        i=0\n        for tr_idx, val_idx in indices:\n            print('encoded fold: ', i)\n            original_indices.extend(val_idx)\n            i+=1\n            cur_val_table = df[val_idx]\n            for cur_column in self.cat_features:\n                cur_grouped_table = df[tr_idx].group_by(cur_column).agg(pl.mean(self.target_col))\n                replace_dict = {row[0] : row[1] for row in cur_grouped_table.iter_rows()}\n\n                mask = ~pl.col(cur_column).is_in(replace_dict.keys())\n\n                cur_val_table = cur_val_table.with_columns(\n                    pl.when(mask)\n                      .then(self.na_value)\n                      .otherwise(pl.col(cur_column))\n                      .alias(cur_column)\n                )\n\n                cur_val_table = cur_val_table.with_columns(\n                    pl.col(cur_column)\n                      .replace(replace_dict)\n                      .cast(pl.Float32)\n                      .alias(cur_column))\n            full_encoded_table.append(cur_val_table)\n        full_encoded_table = pl.concat(full_encoded_table, how='vertical_relaxed')\n        full_encoded_table = full_encoded_table.with_columns(pl.Series(name = 'orig_index', values = original_indices))\n        full_encoded_table = full_encoded_table.sort('orig_index').drop('orig_index')\n        return full_encoded_table","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.477892Z","iopub.execute_input":"2024-04-22T15:13:37.478616Z","iopub.status.idle":"2024-04-22T15:13:37.497924Z","shell.execute_reply.started":"2024-04-22T15:13:37.478588Z","shell.execute_reply":"2024-04-22T15:13:37.496068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cat_encoder = CategoricalEncoder()\n\n# tr_base_merged = cat_encoder.fit_transform_train(tr_base_merged, cat_features=string_cols, target_col='target')","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:13:37.499506Z","iopub.execute_input":"2024-04-22T15:13:37.499955Z","iopub.status.idle":"2024-04-22T15:13:37.515763Z","shell.execute_reply.started":"2024-04-22T15:13:37.499918Z","shell.execute_reply":"2024-04-22T15:13:37.514125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Praca domowa 2024-04-15:\n    - Dodac join do wierszy walidacyjnych z danymi obliczonymi na wierszach treningowych","metadata":{}},{"cell_type":"markdown","source":"### Model training","metadata":{}},{"cell_type":"code","source":"#X = tr_base_merged.select(nonstring_cols + string_cols)\nX = tr_base_merged.select(nonstring_cols)\ny = tr_base_merged['target']","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:26.170936Z","iopub.execute_input":"2024-04-22T15:14:26.171385Z","iopub.status.idle":"2024-04-22T15:14:26.178978Z","shell.execute_reply.started":"2024-04-22T15:14:26.171353Z","shell.execute_reply":"2024-04-22T15:14:26.177576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf_splitter = KFold(n_splits = 5, shuffle = True, random_state = 101)\nindices = []\nfor train_index, valid_index in kf_splitter.split(X):\n    indices.append((train_index, valid_index))\nindices","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:26.812187Z","iopub.execute_input":"2024-04-22T15:14:26.812799Z","iopub.status.idle":"2024-04-22T15:14:27.052794Z","shell.execute_reply.started":"2024-04-22T15:14:26.812630Z","shell.execute_reply":"2024-04-22T15:14:27.051149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'n_estimators': 100, \n         'learning_rate': 0.1, \n         'num_leaves': 31,\n         'min_child_samples': 20,\n         'colsample_bytree': 1.0}\n#https://lightgbm.readthedocs.io/en/latest/pythonapi/lightgbm.LGBMClassifier.html","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:27.216618Z","iopub.execute_input":"2024-04-22T15:14:27.217029Z","iopub.status.idle":"2024-04-22T15:14:27.222628Z","shell.execute_reply.started":"2024-04-22T15:14:27.216998Z","shell.execute_reply":"2024-04-22T15:14:27.220977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Praca domowa 2024-04-22\n - Znalezc optymalne hiperparametry dla modelu i wyslac rozwiazanie na Kaggle","metadata":{}},{"cell_type":"code","source":"models = []\nfor i, (train_index, valid_index) in enumerate(indices):\n    print('Starting Fold ', i)\n    #X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)\n\n    X_train, X_test = X[train_index], X[valid_index]\n    y_train, y_test = y[train_index], y[valid_index]\n\n    classifier = lgb.LGBMClassifier(**params)\n    classifier.fit(X_train, y_train)\n    y_pred = classifier.predict_proba(X_test)[:, 1]\n    competition_metric = 2*roc_auc_score(y_test, y_pred)-1\n    print(competition_metric)\n    models.append(classifier)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:27.650201Z","iopub.execute_input":"2024-04-22T15:14:27.650646Z","iopub.status.idle":"2024-04-22T15:23:19.535321Z","shell.execute_reply.started":"2024-04-22T15:14:27.650616Z","shell.execute_reply":"2024-04-22T15:23:19.533377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.324037Z","iopub.status.idle":"2024-04-22T15:14:02.324926Z","shell.execute_reply.started":"2024-04-22T15:14:02.324598Z","shell.execute_reply":"2024-04-22T15:14:02.324623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_metric\n#to beat 0.5615202870975362","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.326521Z","iopub.status.idle":"2024-04-22T15:14:02.327361Z","shell.execute_reply.started":"2024-04-22T15:14:02.327077Z","shell.execute_reply":"2024-04-22T15:14:02.327101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Predict and submit","metadata":{}},{"cell_type":"code","source":"#test_base_merged = cat_encoder.transform_test(test_base_merged)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.328735Z","iopub.status.idle":"2024-04-22T15:14:02.329403Z","shell.execute_reply.started":"2024-04-22T15:14:02.329193Z","shell.execute_reply":"2024-04-22T15:14:02.329217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_base_merged[X_train.columns]","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.330568Z","iopub.status.idle":"2024-04-22T15:14:02.331212Z","shell.execute_reply.started":"2024-04-22T15:14:02.331006Z","shell.execute_reply":"2024-04-22T15:14:02.331029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encoded = le.fit_transform(X_test.select(string_cols)).astype(np.int16)\n\n# for i, col_name in enumerate(string_cols):\n#     # Replace each column in X with its encoded counterpart\n#     X_test = X_test.with_columns(pl.Series(name=col_name, values=encoded[:, i]))","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.332377Z","iopub.status.idle":"2024-04-22T15:14:02.333008Z","shell.execute_reply.started":"2024-04-22T15:14:02.332811Z","shell.execute_reply":"2024-04-22T15:14:02.332829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_pred = np.array([classifier.predict_proba(X_test)[:, 1] for classifier in models]).mean(axis=0)\nsubmission_pred","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.334187Z","iopub.status.idle":"2024-04-22T15:14:02.334599Z","shell.execute_reply.started":"2024-04-22T15:14:02.334404Z","shell.execute_reply":"2024-04-22T15:14:02.334421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub['score'] = submission_pred\nsample_sub.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.335782Z","iopub.status.idle":"2024-04-22T15:14:02.336179Z","shell.execute_reply.started":"2024-04-22T15:14:02.335992Z","shell.execute_reply":"2024-04-22T15:14:02.336008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accuracy_score(y_pred, y_test)\n# confusion_matrix(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:14:02.337475Z","iopub.status.idle":"2024-04-22T15:14:02.337858Z","shell.execute_reply.started":"2024-04-22T15:14:02.337674Z","shell.execute_reply":"2024-04-22T15:14:02.337689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Praca domowa 2024-04-08:\n    - Add categorical features to the model","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}