{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport warnings\npd.set_option('display.width', 10000)\npd.set_option('display.expand_frame_repr', False)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nwarnings. filterwarnings('ignore') \nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/home-credit-credit-risk-model-stability/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-23T07:38:52.666209Z","iopub.execute_input":"2024-02-23T07:38:52.666570Z","iopub.status.idle":"2024-02-23T07:38:52.982148Z","shell.execute_reply.started":"2024-02-23T07:38:52.666542Z","shell.execute_reply":"2024-02-23T07:38:52.980322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Lets check sample file ######\n###### Also creating pseudo \ndata = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_credit_bureau_a_2_4.csv')\ndata['predict_y'] = np.random.choice([0,1],data.shape[0],p=[0.9,0.1])\nprint(data.shape)\ndata['case_id'] = data['case_id'].astype(object)\ndata['predict_y'] = data['predict_y'].astype(object)\ndata.head(100)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T07:39:16.443833Z","iopub.execute_input":"2024-02-23T07:39:16.444313Z","iopub.status.idle":"2024-02-23T07:40:15.607143Z","shell.execute_reply.started":"2024-02-23T07:39:16.444236Z","shell.execute_reply":"2024-02-23T07:40:15.606272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Getting categorical and numerical column names ######\ncols = data.columns\nnum_cols = data._get_numeric_data().columns\ncat_cols = list(set(cols) - set(num_cols))\nprint(\"num cols are \",num_cols)\nprint(\"cat cols are\",cat_cols )","metadata":{"execution":{"iopub.status.busy":"2024-02-23T07:40:15.608604Z","iopub.execute_input":"2024-02-23T07:40:15.608867Z","iopub.status.idle":"2024-02-23T07:40:15.616038Z","shell.execute_reply.started":"2024-02-23T07:40:15.608842Z","shell.execute_reply":"2024-02-23T07:40:15.614296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### First we check features defination ####\npd.set_option('display.max_colwidth', None)\nfeatures_defination = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv')\nfeatures_defination1 = features_defination[features_defination.Variable.isin(data.columns)]\nfeatures_defination1","metadata":{"execution":{"iopub.status.busy":"2024-02-22T16:30:28.331067Z","iopub.execute_input":"2024-02-22T16:30:28.332733Z","iopub.status.idle":"2024-02-22T16:30:28.387712Z","shell.execute_reply.started":"2024-02-22T16:30:28.332661Z","shell.execute_reply":"2024-02-22T16:30:28.386229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Check data distribution @@@@\ndata[num_cols].describe().T","metadata":{"execution":{"iopub.status.busy":"2024-02-22T06:53:49.925170Z","iopub.execute_input":"2024-02-22T06:53:49.925567Z","iopub.status.idle":"2024-02-22T06:54:05.365698Z","shell.execute_reply.started":"2024-02-22T06:53:49.925536Z","shell.execute_reply":"2024-02-22T06:54:05.364494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Checking unique values \ndata[cat_cols].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-02-22T08:03:02.458760Z","iopub.execute_input":"2024-02-22T08:03:02.459263Z","iopub.status.idle":"2024-02-22T08:03:23.821205Z","shell.execute_reply.started":"2024-02-22T08:03:02.459221Z","shell.execute_reply":"2024-02-22T08:03:23.819927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"####### Check % missing values ####\n(data[num_cols].isnull().sum()/(len(data[num_cols])))*100","metadata":{"execution":{"iopub.status.busy":"2024-02-22T08:03:23.823538Z","iopub.execute_input":"2024-02-22T08:03:23.823890Z","iopub.status.idle":"2024-02-22T08:03:26.182241Z","shell.execute_reply.started":"2024-02-22T08:03:23.823860Z","shell.execute_reply":"2024-02-22T08:03:26.180788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some univaraite analysis","metadata":{}},{"cell_type":"code","source":"########### Distribution of numerical variables #######\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfor col in num_cols:\n    print(col)\n    print('Skew :', round(data[col].skew(), 2))\n    plt.figure(figsize = (15, 4))\n    plt.subplot(1, 2, 1)\n    data[col].hist(grid=False)\n    plt.ylabel('count')\n    plt.subplot(1, 2, 2)\n    sns.boxplot(x=data[col])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-22T08:05:31.661411Z","iopub.execute_input":"2024-02-22T08:05:31.662090Z","iopub.status.idle":"2024-02-22T08:06:06.193066Z","shell.execute_reply.started":"2024-02-22T08:05:31.662052Z","shell.execute_reply":"2024-02-22T08:06:06.191880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"######### Distribution of categorical variables #########\nfig, axes = plt.subplots(3, 2, figsize = (18, 18))\nfig.suptitle('Bar plot for all categorical variables in the dataset')\nsns.countplot(ax = axes[0, 0], x = 'collaterals_typeofguarante_669M', data = data, color = 'blue', \n              order = data['collaterals_typeofguarante_669M'].value_counts().index);\nsns.countplot(ax = axes[0, 1], x = 'subjectroles_name_541M', data = data, color = 'blue', \n              order = data['subjectroles_name_541M'].value_counts().index);\nsns.countplot(ax = axes[1, 0], x = 'collaterals_typeofguarante_359M', data = data, color = 'blue', \n              order = data['collaterals_typeofguarante_359M'].value_counts().index);\nsns.countplot(ax = axes[1, 1], x = 'collater_typofvalofguarant_298M', data = data, color = 'blue', \n              order = data['collater_typofvalofguarant_298M'].value_counts().index);\nsns.countplot(ax = axes[2, 0], x = 'subjectroles_name_838M', data = data, color = 'blue', \n              order = data['subjectroles_name_838M'].head(20).value_counts().index);\nsns.countplot(ax = axes[2, 1], x = 'predict_y', data = data, color = 'blue', \n              order = data['predict_y'].head(20).value_counts().index);\naxes[0][0].tick_params(labelrotation=45);\naxes[0][1].tick_params(labelrotation=45);\naxes[1][0].tick_params(labelrotation=45);\naxes[1][1].tick_params(labelrotation=45);\naxes[2][0].tick_params(labelrotation=45);\naxes[2][1].tick_params(labelrotation=45);","metadata":{"execution":{"iopub.status.busy":"2024-02-22T08:14:06.423585Z","iopub.execute_input":"2024-02-22T08:14:06.424045Z","iopub.status.idle":"2024-02-22T08:16:29.345021Z","shell.execute_reply.started":"2024-02-22T08:14:06.424011Z","shell.execute_reply":"2024-02-22T08:16:29.343845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We will do some multivariate analysis","metadata":{}},{"cell_type":"code","source":"#### Plot correlation martix ######\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 7))\nsns.heatmap(data[num_cols].corr(), annot = True, vmin = -1, vmax = 1)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-22T06:54:26.073836Z","iopub.execute_input":"2024-02-22T06:54:26.074197Z","iopub.status.idle":"2024-02-22T06:54:34.569905Z","shell.execute_reply.started":"2024-02-22T06:54:26.074156Z","shell.execute_reply":"2024-02-22T06:54:34.568349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building","metadata":{}},{"cell_type":"code","source":"list(set(data.columns) - set(['case_id','predict_y']))","metadata":{"execution":{"iopub.status.busy":"2024-02-23T07:40:42.008894Z","iopub.execute_input":"2024-02-23T07:40:42.009335Z","iopub.status.idle":"2024-02-23T07:40:42.017555Z","shell.execute_reply.started":"2024-02-23T07:40:42.009301Z","shell.execute_reply":"2024-02-23T07:40:42.015858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Splitting the data########\nfrom sklearn.model_selection import train_test_split\nX_columns = list(set(data.columns) - set(['case_id','predict_y']))\ny_column = 'predict_y'\nX = data[X_columns]\ny = data['predict_y'].astype('bool')\nfor c in X.columns:\n    col_type = X[c].dtype\n    if col_type == 'object' or col_type.name == 'category':\n        X[c] = X[c].astype('category')\nX_train, X_test, y_train, y_test = train_test_split(X,y ,random_state=104,test_size=0.25,  shuffle=True) ","metadata":{"execution":{"iopub.status.busy":"2024-02-23T07:40:57.383325Z","iopub.execute_input":"2024-02-23T07:40:57.383716Z","iopub.status.idle":"2024-02-23T07:41:18.046062Z","shell.execute_reply.started":"2024-02-23T07:40:57.383689Z","shell.execute_reply":"2024-02-23T07:41:18.044501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y,data\nimport gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-22T16:56:29.094329Z","iopub.execute_input":"2024-02-22T16:56:29.094812Z","iopub.status.idle":"2024-02-22T16:56:29.251350Z","shell.execute_reply.started":"2024-02-22T16:56:29.094755Z","shell.execute_reply":"2024-02-22T16:56:29.249761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###### Lgbm model ########\nimport lightgbm as lgb\nfit_params={\"early_stopping_rounds\":10, \n            \"eval_metric\" : 'auc', \n            \"eval_set\" : [(X_test,y_test)],\n            'eval_names': ['valid'],\n            'verbose': 100,\n            'feature_name': 'auto', # that's actually the default\n            'categorical_feature': 'auto' # that's actually the default\n           }\nclf = lgb.LGBMClassifier(num_leaves= 15, max_depth=-1, \n                         random_state=314, \n                         silent=True, \n                         metric='None', \n                         n_jobs=4, \n                         n_estimators=1000,\n                         colsample_bytree=0.9,\n                         subsample=0.9,\n                         learning_rate=0.1)\n\nclf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-23T07:56:32.161221Z","iopub.execute_input":"2024-02-23T07:56:32.161626Z","iopub.status.idle":"2024-02-23T08:04:12.518332Z","shell.execute_reply.started":"2024-02-23T07:56:32.161595Z","shell.execute_reply":"2024-02-23T08:04:12.516776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_imp = pd.Series(clf.feature_importances_, index=X.columns)\nfeat_imp.nlargest(30).plot(kind='barh', figsize=(8,10))","metadata":{"execution":{"iopub.status.busy":"2024-02-23T08:06:07.923309Z","iopub.execute_input":"2024-02-23T08:06:07.924055Z","iopub.status.idle":"2024-02-23T08:06:08.382221Z","shell.execute_reply.started":"2024-02-23T08:06:07.924004Z","shell.execute_reply":"2024-02-23T08:06:08.381147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nlgb_train = lgb.Dataset(X_train, label=y_train)\nlgb_valid = lgb.Dataset(X_test, label=y_test,reference=lgb_train)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}\n\ngbm = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,\n    callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n)","metadata":{},"execution_count":null,"outputs":[]}]}