{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport gc , re","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-04T13:56:49.136063Z","iopub.execute_input":"2024-04-04T13:56:49.137098Z","iopub.status.idle":"2024-04-04T13:56:50.664118Z","shell.execute_reply.started":"2024-04-04T13:56:49.137044Z","shell.execute_reply":"2024-04-04T13:56:50.662856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_base = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ndf_base.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:56:50.666058Z","iopub.execute_input":"2024-04-04T13:56:50.666529Z","iopub.status.idle":"2024-04-04T13:56:51.150006Z","shell.execute_reply.started":"2024-04-04T13:56:50.666495Z","shell.execute_reply":"2024-04-04T13:56:51.149163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_base.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:56:51.153990Z","iopub.execute_input":"2024-04-04T13:56:51.156361Z","iopub.status.idle":"2024-04-04T13:56:51.441438Z","shell.execute_reply.started":"2024-04-04T13:56:51.156323Z","shell.execute_reply":"2024-04-04T13:56:51.440323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_csv   = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/feature_definitions.csv\")\n# feature_csv.filter(pl.col('Variable') == 'actualdpd_943P')","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:56:51.448158Z","iopub.execute_input":"2024-04-04T13:56:51.451355Z","iopub.status.idle":"2024-04-04T13:56:51.458027Z","shell.execute_reply.started":"2024-04-04T13:56:51.451302Z","shell.execute_reply":"2024-04-04T13:56:51.456925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## data preprocessing\nclass Pipeline :\n    @staticmethod\n    def data_prep():\n        \n        df1  = pl.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv')\n        df1 = df1.with_columns(pl.col('date_decision').cast(pl.Date))\n        df2 =  pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_deposit_1.csv\")\n        df2 = df2.with_columns(pl.col(['openingdate_313D' , 'contractenddate_991D']).cast(pl.Date))\n        df2 = df2.group_by(\"case_id\").agg(pl.col(['amount_416A' , 'num_group1' ]).sum())\n        df3 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_debitcard_1.csv\")\n        df3 = df3.with_columns(pl.col('openingdate_857D').cast(pl.Date))\n        df3 = df3.group_by('case_id').agg(pl.col('num_group1').sum())\n        df1  = df1.join(df2 , on ='case_id' , how ='left')\n        df1 = df1.join(df3 , on ='case_id' , how = 'left')\n        df4  = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_other_1.csv\")\n        df1 = df1.join(df4 , how = 'left' , on = 'case_id' , suffix = \"_\")\n        df5 = pl.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_person_1.csv')\n        df5 = df5[['case_id' , 'birth_259D' , 'education_927M' , 'role_1084L' , 'persontype_1072L' , 'type_25L']]\n        df5 = df5.with_columns(pl.col('birth_259D').cast(pl.Date))\n#         df5 = df5.group_by('case_id').agg(pl.col(['birth_259D' , 'education_927M' , 'role_1084L' , 'persontype_1072L' , 'type_25L']))\n        df5 = df5.group_by('case_id').agg(pl.col(['birth_259D' , 'education_927M' , 'role_1084L' , 'persontype_1072L']))\n        df5 = df5.with_columns(pl.col('persontype_1072L').apply(lambda x : x.sum()))\n        df5 = df5.with_columns(pl.col('birth_259D').list.get(0).cast(pl.Date))\n        df5 = df5.with_columns(pl.col('education_927M').list.get(0))\n        df5 = df5.with_columns(pl.col('role_1084L').list.get(0))\n        df5 = df5.drop('empl_employedfrom_271D')\n        df1 = df1.join(df5 , on ='case_id' , how ='left')\n        df6 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_0.csv\")\n        df7 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_1.csv\")\n        lst = df6.columns\n        matchedLst=[m for m in lst if re.search('date*', m)]\n        df6 = df6.with_columns(pl.col(matchedLst).cast(pl.Date))\n        col = []\n        for i in df6.columns :\n            if df6[i].null_count() < 5 : \n                col.append(i)\n       \n        \n        df1 = df1.join(df6[col] , on ='case_id' , how = 'left')\n        df7 = df7.with_columns(pl.col(matchedLst).cast(pl.Date))\n \n        df1 = df1.join(df7[col] , on ='case_id' , how ='left')\n        df8 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_deposit_1.csv\")\n        df8 = df8.group_by('case_id').agg(pl.col(['amount_416A' , 'openingdate_313D' ]))\n        df8  = df8.with_columns(pl.col('openingdate_313D').list.get(0))\n        df8 = df8.with_columns(pl.col('openingdate_313D').cast(pl.Date))\n        df8 = df8.with_columns(pl.col('amount_416A').apply(lambda x : x.sum()))\n        df1 = df1.join(df8 , how = 'left' , on = 'case_id')\n        \n        \n        \n        \n        \n        \n        \n        \n        \n        \n        \n        \n        \n        \n        del_list = [ df2 , df3 , df4 , df5 , df6 , df7]\n        del del_list\n        \n        return df1\n    @staticmethod\n    def test_data_prep():\n        df1  = pl.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv')\n        df1 = df1.with_columns(pl.col('date_decision').cast(pl.Date))\n        df2 =  pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_deposit_1.csv\")\n        df2 = df2.with_columns(pl.col(['openingdate_313D' , 'contractenddate_991D']).cast(pl.Date))\n        df2 = df2.group_by(\"case_id\").agg(pl.col(['amount_416A' , 'num_group1' ]).sum())\n        df3 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_debitcard_1.csv\")\n        df3 = df3.with_columns(pl.col('openingdate_857D').cast(pl.Date))\n        df3 = df3.group_by('case_id').agg(pl.col('num_group1').sum())\n        df1  = df1.join(df2 , on ='case_id' , how ='left')\n        df1 = df1.join(df3 , on ='case_id' , how = 'left')\n        df4  = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_other_1.csv\")\n        df1 = df1.join(df4 , how = 'left' , on = 'case_id' , suffix = \"_\")\n        df5 = pl.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_person_1.csv')\n        df5 = df5[['case_id' , 'birth_259D' , 'education_927M' , 'role_1084L' , 'persontype_1072L' , 'type_25L']]\n        df5 = df5.with_columns(pl.col('birth_259D').cast(pl.Date))\n#         df5 = df5.group_by('case_id').agg(pl.col(['birth_259D' , 'education_927M' , 'role_1084L' , 'persontype_1072L' , 'type_25L']))\n        df5 = df5.group_by('case_id').agg(pl.col(['birth_259D' , 'education_927M' , 'role_1084L' , 'persontype_1072L']))\n        df5 = df5.with_columns(pl.col('persontype_1072L').apply(lambda x : x.sum()))\n        df5 = df5.with_columns(pl.col('birth_259D').list.get(0).cast(pl.Date))\n        df5 = df5.with_columns(pl.col('education_927M').list.get(0))\n        df5 = df5.with_columns(pl.col('role_1084L').list.get(0))\n        df5 = df5.drop('empl_employedfrom_271D')\n        df1 = df1.join(df5 , on ='case_id' , how ='left')\n        df6 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_0.csv\")\n        df7 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_1.csv\")\n        lst = df6.columns\n        matchedLst=[m for m in lst if re.search('date*', m)]\n        df6 = df6.with_columns(pl.col(matchedLst).cast(pl.Date))\n        col = []\n        for i in df6.columns :\n            if df6[i].null_count() < 5 : \n                col.append(i)\n       \n        \n        df1 = df1.join(df6[col] , on ='case_id' , how = 'left')\n        df7 = df7.with_columns(pl.col(matchedLst).cast(pl.Date))\n \n        df1 = df1.join(df7[col] , on ='case_id' , how ='left')\n        df8 = pl.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_deposit_1.csv\")\n        df8 = df8.group_by('case_id').agg(pl.col(['amount_416A' , 'openingdate_313D' ]))\n        df8  = df8.with_columns(pl.col('openingdate_313D').list.get(0))\n        df8 = df8.with_columns(pl.col('openingdate_313D').cast(pl.Date))\n        df8 = df8.with_columns(pl.col('amount_416A').apply(lambda x : x.sum()))\n        df1 = df1.join(df8 , how = 'left' , on = 'case_id')\n        \n        return df1\n    \n    \n\n        \n        \n\n# df = Pipeline.data_prep()\n# # df = Pipeline.df_join(path3)\n# df.head()        \n                \n                \n        ","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:56:51.460114Z","iopub.execute_input":"2024-04-04T13:56:51.460924Z","iopub.status.idle":"2024-04-04T13:56:51.502478Z","shell.execute_reply.started":"2024-04-04T13:56:51.460880Z","shell.execute_reply":"2024-04-04T13:56:51.501108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_train = Pipeline.data_prep()\ndf_test = Pipeline.test_data_prep()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:56:51.504548Z","iopub.execute_input":"2024-04-04T13:56:51.505830Z","iopub.status.idle":"2024-04-04T13:57:17.261113Z","shell.execute_reply.started":"2024-04-04T13:56:51.505783Z","shell.execute_reply":"2024-04-04T13:57:17.258809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:17.262660Z","iopub.execute_input":"2024-04-04T13:57:17.263258Z","iopub.status.idle":"2024-04-04T13:57:17.285377Z","shell.execute_reply.started":"2024-04-04T13:57:17.263225Z","shell.execute_reply":"2024-04-04T13:57:17.284030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:17.286907Z","iopub.execute_input":"2024-04-04T13:57:17.287258Z","iopub.status.idle":"2024-04-04T13:57:17.334900Z","shell.execute_reply.started":"2024-04-04T13:57:17.287225Z","shell.execute_reply":"2024-04-04T13:57:17.333692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test  = df_test.to_pandas()\ndf_train  =  df_train.to_pandas()\ncase_id = df_test['case_id']\nna_counts = df_test.isna().sum()\ncols_test  = na_counts[na_counts < 4].index\ndf_test  = df_test[cols_test]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:17.336261Z","iopub.execute_input":"2024-04-04T13:57:17.336589Z","iopub.status.idle":"2024-04-04T13:57:18.398257Z","shell.execute_reply.started":"2024-04-04T13:57:17.336560Z","shell.execute_reply":"2024-04-04T13:57:18.397125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:18.404031Z","iopub.execute_input":"2024-04-04T13:57:18.404812Z","iopub.status.idle":"2024-04-04T13:57:19.883063Z","shell.execute_reply.started":"2024-04-04T13:57:18.404765Z","shell.execute_reply":"2024-04-04T13:57:19.881898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_test","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:19.884890Z","iopub.execute_input":"2024-04-04T13:57:19.885703Z","iopub.status.idle":"2024-04-04T13:57:19.894007Z","shell.execute_reply.started":"2024-04-04T13:57:19.885650Z","shell.execute_reply":"2024-04-04T13:57:19.892982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## extracting columns that are present on test as well as training data\ntrain_cols  = df_train.columns\ncols = list(set(cols_test).intersection(train_cols))","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:19.895952Z","iopub.execute_input":"2024-04-04T13:57:19.896378Z","iopub.status.idle":"2024-04-04T13:57:19.904095Z","shell.execute_reply.started":"2024-04-04T13:57:19.896340Z","shell.execute_reply":"2024-04-04T13:57:19.903276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:19.905235Z","iopub.execute_input":"2024-04-04T13:57:19.906021Z","iopub.status.idle":"2024-04-04T13:57:19.917119Z","shell.execute_reply.started":"2024-04-04T13:57:19.905990Z","shell.execute_reply":"2024-04-04T13:57:19.916293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = df_train.target.copy()\ndf_train = df_train[cols]\ndf_train['target'] = target.copy()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:19.918563Z","iopub.execute_input":"2024-04-04T13:57:19.919122Z","iopub.status.idle":"2024-04-04T13:57:20.613915Z","shell.execute_reply.started":"2024-04-04T13:57:19.919086Z","shell.execute_reply":"2024-04-04T13:57:20.612671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dropna(inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:20.615426Z","iopub.execute_input":"2024-04-04T13:57:20.616538Z","iopub.status.idle":"2024-04-04T13:57:21.960729Z","shell.execute_reply.started":"2024-04-04T13:57:20.616497Z","shell.execute_reply":"2024-04-04T13:57:21.959407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_vars  = df_train.select_dtypes('object').columns","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:21.962314Z","iopub.execute_input":"2024-04-04T13:57:21.962773Z","iopub.status.idle":"2024-04-04T13:57:22.131821Z","shell.execute_reply.started":"2024-04-04T13:57:21.962737Z","shell.execute_reply":"2024-04-04T13:57:22.130469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_vars","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:22.133204Z","iopub.execute_input":"2024-04-04T13:57:22.133547Z","iopub.status.idle":"2024-04-04T13:57:22.151847Z","shell.execute_reply.started":"2024-04-04T13:57:22.133520Z","shell.execute_reply":"2024-04-04T13:57:22.150432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# label_encoder = LabelEncoder()\n\n# df_catvars = df_train[cat_vars].apply(label_encoder.fit_transform)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:22.153332Z","iopub.execute_input":"2024-04-04T13:57:22.153785Z","iopub.status.idle":"2024-04-04T13:57:22.161309Z","shell.execute_reply.started":"2024-04-04T13:57:22.153752Z","shell.execute_reply":"2024-04-04T13:57:22.160210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## converting objcets to numeric values\nfrom sklearn.preprocessing import OrdinalEncoder\n\n# Create encoder\nordinal_encoder = OrdinalEncoder(handle_unknown='use_encoded_value',\n                                 unknown_value=-1)\n\ndf_catvars = ordinal_encoder.fit_transform(df_train[cat_vars])\ndf_catvars  = pd.DataFrame( df_catvars ) \ndf_catvars.columns = cat_vars","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:22.162793Z","iopub.execute_input":"2024-04-04T13:57:22.163678Z","iopub.status.idle":"2024-04-04T13:57:25.475981Z","shell.execute_reply.started":"2024-04-04T13:57:22.163644Z","shell.execute_reply":"2024-04-04T13:57:25.474748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns  = cat_vars , inplace = True)\ndf_train  = pd.concat([df_train , df_catvars] , axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:25.477467Z","iopub.execute_input":"2024-04-04T13:57:25.478691Z","iopub.status.idle":"2024-04-04T13:57:26.029990Z","shell.execute_reply.started":"2024-04-04T13:57:25.478644Z","shell.execute_reply":"2024-04-04T13:57:26.028707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['date_decision'] = df_train['date_decision'].astype(int) // 10**9","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.031721Z","iopub.execute_input":"2024-04-04T13:57:26.032134Z","iopub.status.idle":"2024-04-04T13:57:26.044452Z","shell.execute_reply.started":"2024-04-04T13:57:26.032094Z","shell.execute_reply":"2024-04-04T13:57:26.042891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dropna(inplace =  True)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.046185Z","iopub.execute_input":"2024-04-04T13:57:26.046747Z","iopub.status.idle":"2024-04-04T13:57:26.332221Z","shell.execute_reply.started":"2024-04-04T13:57:26.046713Z","shell.execute_reply":"2024-04-04T13:57:26.331300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.334270Z","iopub.execute_input":"2024-04-04T13:57:26.334713Z","iopub.status.idle":"2024-04-04T13:57:26.431108Z","shell.execute_reply.started":"2024-04-04T13:57:26.334670Z","shell.execute_reply":"2024-04-04T13:57:26.429911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols.remove('case_id')","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.432663Z","iopub.execute_input":"2024-04-04T13:57:26.433818Z","iopub.status.idle":"2024-04-04T13:57:26.438233Z","shell.execute_reply.started":"2024-04-04T13:57:26.433776Z","shell.execute_reply":"2024-04-04T13:57:26.437074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import lightgbm as lgb\n# from sklearn.model_selection import GridSearchCV\n\n\n# lgb_classifier = lgb.LGBMClassifier(device='gpu')\n# param_grid = {\n#     'n_estimators': [50 ,100 , 500 ],\n#     'learning_rate': [0.1 ],\n#     'num_leaves': [31, 50, 100] ,\n#     'subsample': [0.8, 0.9 ] \n# }\n\n\n\n# grid_search = GridSearchCV(estimator=lgb_classifier, param_grid=param_grid, cv=5, scoring='accuracy', n_jobs=-1)\n# grid_search.fit(df_train[cols] , df_train.target)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.439477Z","iopub.execute_input":"2024-04-04T13:57:26.439838Z","iopub.status.idle":"2024-04-04T13:57:26.449794Z","shell.execute_reply.started":"2024-04-04T13:57:26.439810Z","shell.execute_reply":"2024-04-04T13:57:26.448683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"Best Parameters:\", grid_search.best_params_)\n# print(\"Best Score:\", grid_search.best_score_)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.451450Z","iopub.execute_input":"2024-04-04T13:57:26.452119Z","iopub.status.idle":"2024-04-04T13:57:26.462336Z","shell.execute_reply.started":"2024-04-04T13:57:26.452063Z","shell.execute_reply":"2024-04-04T13:57:26.461048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrainx , testx , trainy , testy  = train_test_split(df_train[cols] , df_train.target , test_size  = 0.25 , random_state =  0 )\nprint(trainx.shape)\nprint(testx.shape)\nprint(trainy.shape)\nprint(testy.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:26.464688Z","iopub.execute_input":"2024-04-04T13:57:26.465048Z","iopub.status.idle":"2024-04-04T13:57:27.170809Z","shell.execute_reply.started":"2024-04-04T13:57:26.465017Z","shell.execute_reply":"2024-04-04T13:57:27.169506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## model training for lighgbm\nimport lightgbm as lgb\n\ncallbacks  = [\n    lgb.early_stopping(stopping_rounds =  5 , verbose = 1) , \n    lgb.callback.log_evaluation(period  = 5)\n]\n\nlgb_classifier = lgb.LGBMClassifier(n_estimators = 100 , num_leaves = 31 , learning_rate = 0.1 , subsample = 0.8)\nlgb_classifier.fit( trainx , trainy , eval_set = [(testx , testy)] , callbacks =  callbacks )\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:27.177215Z","iopub.execute_input":"2024-04-04T13:57:27.177850Z","iopub.status.idle":"2024-04-04T13:57:37.227005Z","shell.execute_reply.started":"2024-04-04T13:57:27.177815Z","shell.execute_reply":"2024-04-04T13:57:37.225791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## prediction on test data\nresult  = pd.DataFrame({'actual' : testy , 'predicted' : lgb_classifier.predict(testx)} )","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:37.228422Z","iopub.execute_input":"2024-04-04T13:57:37.228866Z","iopub.status.idle":"2024-04-04T13:57:38.180549Z","shell.execute_reply.started":"2024-04-04T13:57:37.228825Z","shell.execute_reply":"2024-04-04T13:57:38.179432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Evaluation on test data\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report , roc_curve, roc_auc_score\nacc = accuracy_score(result.actual , result.predicted)\nprint(\"Accuracy score \" ,acc)\nprint()\n\ncm = confusion_matrix(result.actual , result.predicted)\nprint(\"Confusion Matrix:\")\nprint(cm)\nprint()\n\n\nreport = classification_report(result.actual , result.predicted)\nprint(\"Classification Report:\" , report)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:38.181934Z","iopub.execute_input":"2024-04-04T13:57:38.182354Z","iopub.status.idle":"2024-04-04T13:57:38.955554Z","shell.execute_reply.started":"2024-04-04T13:57:38.182316Z","shell.execute_reply":"2024-04-04T13:57:38.954231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## calc. auc score and probability of prediction \nimport matplotlib.pyplot as plt\n\ny_pred_prob  = lgb_classifier.predict_proba(testx)[:, 1]\nfpr , tpr , thresholds  = roc_curve(result.actual , y_pred_prob , pos_label =  1)\nauc_score = roc_auc_score(result.actual , y_pred_prob)\n\n\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, label='ROC Curve (AUC = {:.2f})'.format(auc_score) )\nplt.plot([0, 1], [0, 1], linestyle='--', color='gray', label='Random')\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve for LightGBM Model')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:38.959086Z","iopub.execute_input":"2024-04-04T13:57:38.959452Z","iopub.status.idle":"2024-04-04T13:57:40.343790Z","shell.execute_reply.started":"2024-04-04T13:57:38.959423Z","shell.execute_reply":"2024-04-04T13:57:40.342685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## cutt off \npredicted_proba=  lgb_classifier.predict_proba(testx)[: , 1]\noptimal_proba_cutoff = sorted(list(zip(np.abs(tpr - fpr), thresholds)), key=lambda i: i[0], reverse=True)[0][1]\nroc_predictions = [1 if i >= optimal_proba_cutoff else 0 for i in predicted_proba ]\nresult['predicted'] = roc_predictions","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:40.345180Z","iopub.execute_input":"2024-04-04T13:57:40.346196Z","iopub.status.idle":"2024-04-04T13:57:41.380535Z","shell.execute_reply.started":"2024-04-04T13:57:40.346151Z","shell.execute_reply":"2024-04-04T13:57:41.379265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## metrics\nacc = accuracy_score(result.actual , result.predicted)\nprint(\"Accuracy score \" ,acc)\nprint()\n\ncm = confusion_matrix(result.actual , result.predicted)\nprint(\"Confusion Matrix:\")\nprint(cm)\nprint()\n\n\nreport = classification_report(result.actual , result.predicted)\nreport\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:41.381819Z","iopub.execute_input":"2024-04-04T13:57:41.382398Z","iopub.status.idle":"2024-04-04T13:57:42.140309Z","shell.execute_reply.started":"2024-04-04T13:57:41.382364Z","shell.execute_reply":"2024-04-04T13:57:42.139364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[cat_vars] = df_test[cat_vars].apply(lambda x: x.fillna(x.mode()[0]))\ndf_catvars = ordinal_encoder.transform(df_test[cat_vars].astype(\"O\"))\ndf_catvars = pd.DataFrame(df_catvars)\ndf_catvars.columns = cat_vars","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:42.141374Z","iopub.execute_input":"2024-04-04T13:57:42.141676Z","iopub.status.idle":"2024-04-04T13:57:42.164378Z","shell.execute_reply.started":"2024-04-04T13:57:42.141652Z","shell.execute_reply":"2024-04-04T13:57:42.163159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf_test.drop(columns  = cat_vars , inplace = True)\ndf_test  = pd.concat([df_test , df_catvars ] , axis = 1)\ndf_test['date_decision'] = df_test['date_decision'].astype(int) // 10**9\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:42.165908Z","iopub.execute_input":"2024-04-04T13:57:42.166466Z","iopub.status.idle":"2024-04-04T13:57:42.174927Z","shell.execute_reply.started":"2024-04-04T13:57:42.166430Z","shell.execute_reply":"2024-04-04T13:57:42.173461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_proba  = lgb_classifier.predict_proba(df_test[cols])[:,1]\nroc_predictions = [1 if i >= optimal_proba_cutoff else 0 for i in predicted_proba ]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:42.176149Z","iopub.execute_input":"2024-04-04T13:57:42.176434Z","iopub.status.idle":"2024-04-04T13:57:42.192120Z","shell.execute_reply.started":"2024-04-04T13:57:42.176409Z","shell.execute_reply":"2024-04-04T13:57:42.191032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## submission \nsubmission  = pd.DataFrame({'case_id' : case_id , 'score' : predicted_proba})\nsubmission.to_csv(\"/kaggle/working/submission.csv\" , index  = False)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T13:57:42.193471Z","iopub.execute_input":"2024-04-04T13:57:42.193804Z","iopub.status.idle":"2024-04-04T13:57:42.206868Z","shell.execute_reply.started":"2024-04-04T13:57:42.193778Z","shell.execute_reply":"2024-04-04T13:57:42.205693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}