{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-07T08:28:00.468481Z","iopub.execute_input":"2024-02-07T08:28:00.470005Z","iopub.status.idle":"2024-02-07T08:28:03.699343Z","shell.execute_reply.started":"2024-02-07T08:28:00.469939Z","shell.execute_reply":"2024-02-07T08:28:03.697746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Depth 0 Features ","metadata":{}},{"cell_type":"code","source":"train_base = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_base.parquet')\ntrain_base.set_index('case_id')\n\ntrain_static_0 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_0_0.parquet')\ntrain_static_1 = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_0_1.parquet')\ntrain_static = pd.concat([train_static_0, train_static_1])\ntrain_static = train_static.set_index('case_id')\n\ntrain_static_cb = pd.read_parquet('/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train/train_static_cb_0.parquet')\ntrain_static_cb = train_static_cb.set_index('case_id')","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:31:31.899002Z","iopub.execute_input":"2024-02-07T08:31:31.899523Z","iopub.status.idle":"2024-02-07T08:31:46.831391Z","shell.execute_reply.started":"2024-02-07T08:31:31.899485Z","shell.execute_reply":"2024-02-07T08:31:46.830046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train  = pd.concat([train_base, train_static, train_static_cb], axis=1)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:31:46.833328Z","iopub.execute_input":"2024-02-07T08:31:46.833724Z","iopub.status.idle":"2024-02-07T08:31:52.693967Z","shell.execute_reply.started":"2024-02-07T08:31:46.833691Z","shell.execute_reply":"2024-02-07T08:31:52.692180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_base, train_static, train_static_0, train_static_1, train_static_cb","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:31:52.696442Z","iopub.execute_input":"2024-02-07T08:31:52.698386Z","iopub.status.idle":"2024-02-07T08:31:53.263740Z","shell.execute_reply.started":"2024-02-07T08:31:52.698319Z","shell.execute_reply":"2024-02-07T08:31:53.262314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reomove Columns and Rows with high Null value fraction","metadata":{}},{"cell_type":"code","source":"colwise_null = (train.isna().sum(axis=0) / len(train)).sort_values(ascending=True)\ncolwise_null.plot.hist(bins=100)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:31:53.267511Z","iopub.execute_input":"2024-02-07T08:31:53.268612Z","iopub.status.idle":"2024-02-07T08:32:01.687942Z","shell.execute_reply.started":"2024-02-07T08:31:53.268572Z","shell.execute_reply":"2024-02-07T08:32:01.686407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(colwise_null[colwise_null > 0.5].index, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:01.689391Z","iopub.execute_input":"2024-02-07T08:32:01.689789Z","iopub.status.idle":"2024-02-07T08:32:03.190258Z","shell.execute_reply.started":"2024-02-07T08:32:01.689756Z","shell.execute_reply":"2024-02-07T08:32:03.188732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rowwise_null = (train.isna().sum(axis=1)/ len(train.columns)).sort_values(ascending=True)\nrowwise_null.plot.hist(bins=100)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:03.191410Z","iopub.execute_input":"2024-02-07T08:32:03.191780Z","iopub.status.idle":"2024-02-07T08:32:10.117279Z","shell.execute_reply.started":"2024-02-07T08:32:03.191749Z","shell.execute_reply":"2024-02-07T08:32:10.115761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(rowwise_null[rowwise_null > 0.5].index)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:10.118858Z","iopub.execute_input":"2024-02-07T08:32:10.119263Z","iopub.status.idle":"2024-02-07T08:32:12.079523Z","shell.execute_reply.started":"2024-02-07T08:32:10.119229Z","shell.execute_reply":"2024-02-07T08:32:12.077924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **removing every row that has a null in target cols** ","metadata":{}},{"cell_type":"code","source":"train = train.drop(train[train['target'].isna()].index)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:12.081892Z","iopub.execute_input":"2024-02-07T08:32:12.082459Z","iopub.status.idle":"2024-02-07T08:32:13.509399Z","shell.execute_reply.started":"2024-02-07T08:32:12.082410Z","shell.execute_reply":"2024-02-07T08:32:13.507765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:13.511225Z","iopub.execute_input":"2024-02-07T08:32:13.511883Z","iopub.status.idle":"2024-02-07T08:32:13.519008Z","shell.execute_reply.started":"2024-02-07T08:32:13.511828Z","shell.execute_reply":"2024-02-07T08:32:13.517717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'] = train['target'].astype(np.int16)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:13.522894Z","iopub.execute_input":"2024-02-07T08:32:13.523502Z","iopub.status.idle":"2024-02-07T08:32:13.535367Z","shell.execute_reply.started":"2024-02-07T08:32:13.523457Z","shell.execute_reply":"2024-02-07T08:32:13.534053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_features  = train.dtypes[(train.dtypes == np.number)].index\ndatetime_features = ['dateofbirth_337D', 'lastapprdate_640D', 'lastapplicationdate_877D', 'lastactivateddate_801D', 'date_decision']\ntarget = 'target'\ncat_features = train.columns.drop(set(list(num_features) + list(datetime_features) + ['target']))","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:13.537048Z","iopub.execute_input":"2024-02-07T08:32:13.537455Z","iopub.status.idle":"2024-02-07T08:32:13.548782Z","shell.execute_reply.started":"2024-02-07T08:32:13.537414Z","shell.execute_reply":"2024-02-07T08:32:13.547382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Datetime Feature to int Features","metadata":{}},{"cell_type":"code","source":"for feature in datetime_features:\n    train[feature] = pd.to_datetime(train[feature])\n    \n    train[feature+'_'+'year'] = train[feature].dt.year\n    train[feature+'_'+'month'] = train[feature].dt.month\n    train[feature+'_'+'day'] = train[feature].dt.day\n    \n    train = train.drop(feature, axis=1)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:32:21.344965Z","iopub.execute_input":"2024-02-07T08:32:21.345381Z","iopub.status.idle":"2024-02-07T08:32:25.813836Z","shell.execute_reply.started":"2024-02-07T08:32:21.345350Z","shell.execute_reply":"2024-02-07T08:32:25.812750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Encoding Categorical features","metadata":{}},{"cell_type":"code","source":"#replace nan with -1\ntrain[cat_features].fillna(-1)\nfeature_maps = {}\n\nfor feature in cat_features:\n    feature_maps[feature] = {value: i for i, value in enumerate(train[feature].unique())}\n    feature_maps[feature][-1] = -1\n    \n    train[feature] = train[feature].map(feature_maps[feature])\n    \nfeature_maps    ","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:22.587357Z","iopub.execute_input":"2024-02-07T08:33:22.589790Z","iopub.status.idle":"2024-02-07T08:33:29.308671Z","shell.execute_reply.started":"2024-02-07T08:33:22.589740Z","shell.execute_reply":"2024-02-07T08:33:29.307752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[cat_features[0]]","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:29.310531Z","iopub.execute_input":"2024-02-07T08:33:29.311114Z","iopub.status.idle":"2024-02-07T08:33:29.319736Z","shell.execute_reply.started":"2024-02-07T08:33:29.311080Z","shell.execute_reply":"2024-02-07T08:33:29.318523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:30.269153Z","iopub.execute_input":"2024-02-07T08:33:30.270216Z","iopub.status.idle":"2024-02-07T08:33:30.299146Z","shell.execute_reply.started":"2024-02-07T08:33:30.270178Z","shell.execute_reply":"2024-02-07T08:33:30.298320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_copy = train.copy()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:32.784103Z","iopub.execute_input":"2024-02-07T08:33:32.784515Z","iopub.status.idle":"2024-02-07T08:33:34.296726Z","shell.execute_reply.started":"2024-02-07T08:33:32.784486Z","shell.execute_reply":"2024-02-07T08:33:34.295212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split into Train and Test","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain, test = train_test_split(train, test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:36.824592Z","iopub.execute_input":"2024-02-07T08:33:36.826186Z","iopub.status.idle":"2024-02-07T08:33:39.463873Z","shell.execute_reply.started":"2024-02-07T08:33:36.826133Z","shell.execute_reply":"2024-02-07T08:33:39.462327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Remove Null Values","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer()\ntrain[train.columns.drop('target')] = imputer.fit_transform(train[train.columns.drop('target')])\ntest[test.columns.drop('target')] = imputer.transform(test[test.columns.drop('target')])","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:44.907810Z","iopub.execute_input":"2024-02-07T08:33:44.908359Z","iopub.status.idle":"2024-02-07T08:33:50.593181Z","shell.execute_reply.started":"2024-02-07T08:33:44.908320Z","shell.execute_reply":"2024-02-07T08:33:50.591775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scaling featueres","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\ntrain[train.columns.drop('target')] = scaler.fit_transform(train[train.columns.drop('target')])\ntest[test.columns.drop('target')] = scaler.transform(test[test.columns.drop('target')])","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:52.005287Z","iopub.execute_input":"2024-02-07T08:33:52.005791Z","iopub.status.idle":"2024-02-07T08:33:55.266959Z","shell.execute_reply.started":"2024-02-07T08:33:52.005756Z","shell.execute_reply":"2024-02-07T08:33:55.265505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:33:56.604025Z","iopub.execute_input":"2024-02-07T08:33:56.604463Z","iopub.status.idle":"2024-02-07T08:33:56.799737Z","shell.execute_reply.started":"2024-02-07T08:33:56.604433Z","shell.execute_reply":"2024-02-07T08:33:56.798575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## resample to balance dataset","metadata":{}},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\nX_train, y_train = train.drop('target', axis=1), train['target']\nX_test, y_test = test.drop('target', axis=1), test['target']\n\nsampler = SMOTE()\nX_train, y_train = sampler.fit_resample(X_train, y_train)\n#test[test.columns.drop('target')] = imputer.transform(test[test.columns.drop('target')])","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:34:03.520111Z","iopub.execute_input":"2024-02-07T08:34:03.520749Z","iopub.status.idle":"2024-02-07T08:34:15.758039Z","shell.execute_reply.started":"2024-02-07T08:34:03.520702Z","shell.execute_reply":"2024-02-07T08:34:15.756774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:34:17.108818Z","iopub.execute_input":"2024-02-07T08:34:17.109465Z","iopub.status.idle":"2024-02-07T08:34:17.316387Z","shell.execute_reply.started":"2024-02-07T08:34:17.109430Z","shell.execute_reply":"2024-02-07T08:34:17.314777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling with LGBMClassifier","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.metrics import roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:34:22.519149Z","iopub.execute_input":"2024-02-07T08:34:22.519553Z","iopub.status.idle":"2024-02-07T08:34:24.296072Z","shell.execute_reply.started":"2024-02-07T08:34:22.519523Z","shell.execute_reply":"2024-02-07T08:34:24.294913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.drop(['case_id', 'MONTH', 'WEEK_NUM'], axis=1)\nX_test = X_test.drop(['case_id', 'MONTH', 'WEEK_NUM'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:34:24.333483Z","iopub.execute_input":"2024-02-07T08:34:24.333961Z","iopub.status.idle":"2024-02-07T08:34:25.152870Z","shell.execute_reply.started":"2024-02-07T08:34:24.333927Z","shell.execute_reply":"2024-02-07T08:34:25.151661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LGBMClassifier(n_estimators=50)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:34:29.144898Z","iopub.execute_input":"2024-02-07T08:34:29.145355Z","iopub.status.idle":"2024-02-07T08:35:14.185495Z","shell.execute_reply.started":"2024-02-07T08:34:29.145323Z","shell.execute_reply":"2024-02-07T08:35:14.184028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_prob = model.predict_proba(X_train)[:, 1]\nprint(roc_auc_score(y_train, train_prob))","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:35:14.187364Z","iopub.execute_input":"2024-02-07T08:35:14.187843Z","iopub.status.idle":"2024-02-07T08:35:23.735121Z","shell.execute_reply.started":"2024-02-07T08:35:14.187807Z","shell.execute_reply":"2024-02-07T08:35:23.734172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prob = model.predict_proba(X_test)[:, 1]\nprint(roc_auc_score(y_test, test_prob))","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:35:23.736522Z","iopub.execute_input":"2024-02-07T08:35:23.737074Z","iopub.status.idle":"2024-02-07T08:35:25.879988Z","shell.execute_reply.started":"2024-02-07T08:35:23.737043Z","shell.execute_reply":"2024-02-07T08:35:25.878754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\npred = model.predict(X_train)\nprint(classification_report(y_train, pred))\n\npred = model.predict(X_test)\nprint(classification_report(y_test, pred))","metadata":{"execution":{"iopub.status.busy":"2024-02-07T08:35:25.884915Z","iopub.execute_input":"2024-02-07T08:35:25.885364Z","iopub.status.idle":"2024-02-07T08:35:39.011660Z","shell.execute_reply.started":"2024-02-07T08:35:25.885330Z","shell.execute_reply":"2024-02-07T08:35:39.010120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparam_grid = {\n    'boosting_type': ['gbdt', 'dart', 'rf'],\n    'max_depth': [1, 2, 4, 8],\n    'n_estimators': [50, 100, 1000],\n    'learning_rate': [0.01, 0.05, 0.1, 0.5],\n    'reg_alpha': [0.1, 0.4],\n    'reg_lambda':[0.1, 0.4]\n    \n}\n\nclf = GridSearchCV(model, param_grid)\nclf.fit(X_train, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.best_params","metadata":{},"execution_count":null,"outputs":[]}]}