{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport gc\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-16T05:28:55.026919Z","iopub.execute_input":"2023-01-16T05:28:55.027910Z","iopub.status.idle":"2023-01-16T05:28:56.805544Z","shell.execute_reply.started":"2023-01-16T05:28:55.027784Z","shell.execute_reply":"2023-01-16T05:28:56.804453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset_ = pd.read_feather('/kaggle/input/amexfeather/train_data.ftr')\n# Keep the latest statement features for each customer\ntrain_dataset = train_dataset_.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:29:02.667978Z","iopub.execute_input":"2023-01-16T05:29:02.668359Z","iopub.status.idle":"2023-01-16T05:29:28.336675Z","shell.execute_reply.started":"2023-01-16T05:29:02.668326Z","shell.execute_reply":"2023-01-16T05:29:28.335584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dataset_\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:30:31.340250Z","iopub.execute_input":"2023-01-16T05:30:31.340630Z","iopub.status.idle":"2023-01-16T05:30:31.467123Z","shell.execute_reply.started":"2023-01-16T05:30:31.340596Z","shell.execute_reply":"2023-01-16T05:30:31.466091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:30:35.090536Z","iopub.execute_input":"2023-01-16T05:30:35.090906Z","iopub.status.idle":"2023-01-16T05:30:35.122938Z","shell.execute_reply.started":"2023-01-16T05:30:35.090873Z","shell.execute_reply":"2023-01-16T05:30:35.121938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"train_dataset.info(max_cols=191,show_counts=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:30:39.212295Z","iopub.execute_input":"2023-01-16T05:30:39.212671Z","iopub.status.idle":"2023-01-16T05:30:39.576842Z","shell.execute_reply.started":"2023-01-16T05:30:39.212639Z","shell.execute_reply":"2023-01-16T05:30:39.575772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nnum_cols = [col for col in train_dataset.columns if col not in categorical_cols + [\"target\"]]\n\nprint(f'Total number of features: {1}')\nprint(f'Total number of categorical features: {len(categorical_cols)}')\nprint(f'Total number of continuos features: {len(num_cols)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:30:47.300929Z","iopub.execute_input":"2023-01-16T05:30:47.301465Z","iopub.status.idle":"2023-01-16T05:30:47.309190Z","shell.execute_reply.started":"2023-01-16T05:30:47.301429Z","shell.execute_reply":"2023-01-16T05:30:47.307807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Imputation","metadata":{}},{"cell_type":"code","source":"NaN_Val = np.array(train_dataset.isnull().sum())\nNaN_prec = np.array((train_dataset.isnull().sum() * 100 / len(train_dataset)).round(2))\nNaN_Col = pd.DataFrame([np.array(list(train_dataset.columns)).T,NaN_Val.T,NaN_prec.T,np.array(list(train_dataset.dtypes)).T], index=['Features','Num of Missing values','Percentage','DataType']\n).transpose()\npd.set_option('display.max_rows', None)\nNaN_Col","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:30:50.402017Z","iopub.execute_input":"2023-01-16T05:30:50.402815Z","iopub.status.idle":"2023-01-16T05:30:51.055786Z","shell.execute_reply.started":"2023-01-16T05:30:50.402777Z","shell.execute_reply":"2023-01-16T05:30:51.054230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Remove columns if there are > 80% of missing values as they will affect to create outliers**","metadata":{}},{"cell_type":"code","source":"train_dataset = train_dataset.drop(['S_2','D_66','D_42','D_49','D_73','D_76','R_9','B_29','D_87','D_88','D_106','R_26','D_108','D_110','D_111','B_39','B_42','D_132','D_134','D_135','D_136','D_137','D_138','D_142'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:00.883517Z","iopub.execute_input":"2023-01-16T05:31:00.883913Z","iopub.status.idle":"2023-01-16T05:31:01.135659Z","shell.execute_reply.started":"2023-01-16T05:31:00.883878Z","shell.execute_reply":"2023-01-16T05:31:01.134600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fill Null values","metadata":{}},{"cell_type":"code","source":"selected_col = np.array(['P_2','S_3','B_2','D_41','D_43','B_3','D_44','D_45','D_46','D_48','D_50','D_53','S_7','D_56','S_9','B_6','B_8','D_52','P_3','D_54','D_55','B_13','D_59','D_61','B_15','D_62','B_16','B_17','D_77','B_19','B_20','D_69','B_22','D_70','D_72','D_74','R_7','B_25','B_26','D_78','D_79','D_80','B_27','D_81','R_12','D_82','D_105','S_27','D_83','R_14','D_84','D_86','R_20','B_33','D_89','D_91','S_22','S_23','S_24','S_25','S_26','D_102','D_103','D_104','D_107','B_37','R_27','D_109','D_112','B_40','D_113','D_115','D_118','D_119','D_121','D_122','D_123','D_124','D_125','D_128','D_129','B_41','D_130','D_131','D_133','D_139','D_140','D_141','D_143','D_144','D_145'])\n\nfor col in selected_col:\n    train_dataset[col] = train_dataset[col].fillna(train_dataset[col].median())","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:04.684604Z","iopub.execute_input":"2023-01-16T05:31:04.684984Z","iopub.status.idle":"2023-01-16T05:31:05.855163Z","shell.execute_reply.started":"2023-01-16T05:31:04.684953Z","shell.execute_reply":"2023-01-16T05:31:05.854164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selcted_col2 = np.array(['D_68','B_30','B_38','D_64','D_114','D_116','D_117','D_120','D_126'])\n\nfor col2 in selcted_col2:\n    train_dataset[col2] =  train_dataset[col2].fillna(train_dataset[col2].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:09.350056Z","iopub.execute_input":"2023-01-16T05:31:09.350787Z","iopub.status.idle":"2023-01-16T05:31:09.401667Z","shell.execute_reply.started":"2023-01-16T05:31:09.350732Z","shell.execute_reply":"2023-01-16T05:31:09.400717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_dataset.isnull().sum().to_string())","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:12.283168Z","iopub.execute_input":"2023-01-16T05:31:12.283864Z","iopub.status.idle":"2023-01-16T05:31:12.526419Z","shell.execute_reply.started":"2023-01-16T05:31:12.283828Z","shell.execute_reply":"2023-01-16T05:31:12.525381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:18.513475Z","iopub.execute_input":"2023-01-16T05:31:18.513919Z","iopub.status.idle":"2023-01-16T05:31:18.525079Z","shell.execute_reply.started":"2023-01-16T05:31:18.513878Z","shell.execute_reply":"2023-01-16T05:31:18.524090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:21.260618Z","iopub.execute_input":"2023-01-16T05:31:21.261145Z","iopub.status.idle":"2023-01-16T05:31:21.298447Z","shell.execute_reply.started":"2023-01-16T05:31:21.261095Z","shell.execute_reply":"2023-01-16T05:31:21.297383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Dataset","metadata":{}},{"cell_type":"code","source":"test_dataset_ = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')\n# Keep the latest statement features for each customer\ntest_dataset = test_dataset_.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:31:25.042230Z","iopub.execute_input":"2023-01-16T05:31:25.042627Z","iopub.status.idle":"2023-01-16T05:32:24.944088Z","shell.execute_reply.started":"2023-01-16T05:31:25.042597Z","shell.execute_reply":"2023-01-16T05:32:24.942988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dataset_\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:30.100008Z","iopub.execute_input":"2023-01-16T05:32:30.100398Z","iopub.status.idle":"2023-01-16T05:32:30.247910Z","shell.execute_reply.started":"2023-01-16T05:32:30.100363Z","shell.execute_reply":"2023-01-16T05:32:30.246706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:33.172036Z","iopub.execute_input":"2023-01-16T05:32:33.172785Z","iopub.status.idle":"2023-01-16T05:32:33.200040Z","shell.execute_reply.started":"2023-01-16T05:32:33.172744Z","shell.execute_reply":"2023-01-16T05:32:33.199032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:36.600438Z","iopub.execute_input":"2023-01-16T05:32:36.601131Z","iopub.status.idle":"2023-01-16T05:32:36.608237Z","shell.execute_reply.started":"2023-01-16T05:32:36.601085Z","shell.execute_reply":"2023-01-16T05:32:36.607028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NaN_Val2 = np.array(test_dataset.isnull().sum())\nNaN_prec2 = np.array((test_dataset.isnull().sum() * 100 / len(test_dataset)).round(2))\nNaN_Col2 = pd.DataFrame([np.array(list(test_dataset.columns)).T,NaN_Val2.T,NaN_prec2.T,np.array(list(test_dataset.dtypes)).T], index=['Features','Num of Missing values','Percentage','DataType']\n).transpose()\npd.set_option('display.max_rows', None)\n\nNaN_Col2","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:39.633287Z","iopub.execute_input":"2023-01-16T05:32:39.634483Z","iopub.status.idle":"2023-01-16T05:32:40.903640Z","shell.execute_reply.started":"2023-01-16T05:32:39.634442Z","shell.execute_reply":"2023-01-16T05:32:40.902486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = test_dataset.drop(['S_2','D_42','D_49','D_66','D_73','D_76','R_9','B_29','D_87','D_88','D_106','R_26','D_108','D_110','D_111','B_39','B_42','D_132','D_134','D_135','D_136','D_137','D_138','D_142'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:50.173686Z","iopub.execute_input":"2023-01-16T05:32:50.174031Z","iopub.status.idle":"2023-01-16T05:32:50.668386Z","shell.execute_reply.started":"2023-01-16T05:32:50.174001Z","shell.execute_reply":"2023-01-16T05:32:50.667404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_column = np.array(['P_2','S_3','B_2','D_41','D_43','B_3','D_44','D_45','D_46','D_48','D_50','D_53','S_7','D_56','S_9','S_12','S_17','B_6','B_8','D_52','P_3','D_54','D_55','B_13','D_59','D_61','B_15','D_62','B_16','B_17','D_77','B_19','B_20','D_69','B_22','D_70','D_72','D_74','R_7','B_25','B_26','D_78','D_79','D_80','B_27','D_81','R_12','D_82','D_105','S_27','D_83','R_14','D_84','D_86','R_20','B_33','D_89','D_91','S_22','S_23','S_24','S_25','S_26','D_102','D_103','D_104','D_107','B_37','R_27','D_109','D_112','B_40','D_113','D_115','D_118','D_119','D_121','D_122','D_123','D_124','D_125','D_128','D_129','B_41','D_130','D_131','D_133','D_139','D_140','D_141','D_143','D_144','D_145'])\n\nfor column in selected_column:\n    test_dataset[column] = test_dataset[column].fillna(test_dataset[column].median())","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:53.220690Z","iopub.execute_input":"2023-01-16T05:32:53.221738Z","iopub.status.idle":"2023-01-16T05:32:55.480647Z","shell.execute_reply.started":"2023-01-16T05:32:53.221678Z","shell.execute_reply":"2023-01-16T05:32:55.479621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_column2 = np.array(['D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126'])\n\nfor column2 in selected_column2:\n    test_dataset[column2] =  test_dataset[column2].fillna(test_dataset[column2].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:32:58.141135Z","iopub.execute_input":"2023-01-16T05:32:58.141890Z","iopub.status.idle":"2023-01-16T05:32:58.214734Z","shell.execute_reply.started":"2023-01-16T05:32:58.141846Z","shell.execute_reply":"2023-01-16T05:32:58.213569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_dataset.isnull().sum().to_string())","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:07.771083Z","iopub.execute_input":"2023-01-16T05:33:07.771826Z","iopub.status.idle":"2023-01-16T05:33:08.259462Z","shell.execute_reply.started":"2023-01-16T05:33:07.771789Z","shell.execute_reply":"2023-01-16T05:33:08.258415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:15.351724Z","iopub.execute_input":"2023-01-16T05:33:15.352243Z","iopub.status.idle":"2023-01-16T05:33:15.364092Z","shell.execute_reply.started":"2023-01-16T05:33:15.352155Z","shell.execute_reply":"2023-01-16T05:33:15.362988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:18.330701Z","iopub.execute_input":"2023-01-16T05:33:18.331126Z","iopub.status.idle":"2023-01-16T05:33:18.370155Z","shell.execute_reply.started":"2023-01-16T05:33:18.331091Z","shell.execute_reply":"2023-01-16T05:33:18.369337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical Encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nenc = OrdinalEncoder()\ncategorical_cols.remove('D_66')\n\ntrain_dataset[categorical_cols] = enc.fit_transform(train_dataset[categorical_cols])\ntest_dataset[categorical_cols] = enc.transform(test_dataset[categorical_cols])","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:39.391230Z","iopub.execute_input":"2023-01-16T05:33:39.391603Z","iopub.status.idle":"2023-01-16T05:33:41.111791Z","shell.execute_reply.started":"2023-01-16T05:33:39.391572Z","shell.execute_reply":"2023-01-16T05:33:41.110725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:44.672856Z","iopub.execute_input":"2023-01-16T05:33:44.673286Z","iopub.status.idle":"2023-01-16T05:33:44.680620Z","shell.execute_reply.started":"2023-01-16T05:33:44.673238Z","shell.execute_reply":"2023-01-16T05:33:44.679310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_columns = [col for col in train_dataset.columns if col not in [\"target\"]]\n\nX = train_dataset[num_columns]\ny = train_dataset['target']\n\nprint(f\"X shape is = {X.shape}\" )\nprint(f\"Y shape is = {y.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:50.721597Z","iopub.execute_input":"2023-01-16T05:33:50.721992Z","iopub.status.idle":"2023-01-16T05:33:51.004844Z","shell.execute_reply.started":"2023-01-16T05:33:50.721962Z","shell.execute_reply":"2023-01-16T05:33:51.003776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train-Test Split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nx_train,x_test,y_train,y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\nprint(f\"X_train shape is = {x_train.shape}\" )\nprint(f\"Y_train shape is = {y_train.shape}\" )\nprint(f\"X_test shape is = {x_test.shape}\" )\nprint(f\"Y_test shape is = {y_test.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:53.583053Z","iopub.execute_input":"2023-01-16T05:33:53.583442Z","iopub.status.idle":"2023-01-16T05:33:54.836835Z","shell.execute_reply.started":"2023-01-16T05:33:53.583408Z","shell.execute_reply":"2023-01-16T05:33:54.835652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lightgbm","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\n\nd_train = lgb.Dataset(x_train, label=y_train, categorical_feature = categorical_cols)\n\nparams = {'objective': 'binary','n_estimators': 1400,'metric': 'binary_logloss','boosting': 'gbdt','num_leaves': 90,'reg_lambda' : 50,'colsample_bytree': 0.19,'learning_rate': 0.03,'min_child_samples': 2400,'max_bins': 511,'seed': 42,'verbose': -1}\n\n# trained model with 100 iterations\nmodel = lgb.train(params, d_train, 100)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:33:58.850102Z","iopub.execute_input":"2023-01-16T05:33:58.850483Z","iopub.status.idle":"2023-01-16T05:37:20.847361Z","shell.execute_reply.started":"2023-01-16T05:33:58.850450Z","shell.execute_reply":"2023-01-16T05:37:20.846495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lightgbm Validation","metadata":{}},{"cell_type":"code","source":"valid_predict = model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:40:51.560703Z","iopub.execute_input":"2023-01-16T05:40:51.561110Z","iopub.status.idle":"2023-01-16T05:41:10.443356Z","shell.execute_reply.started":"2023-01-16T05:40:51.561075Z","shell.execute_reply":"2023-01-16T05:41:10.442509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_predict","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:41:26.056967Z","iopub.execute_input":"2023-01-16T05:41:26.057355Z","iopub.status.idle":"2023-01-16T05:41:26.065133Z","shell.execute_reply.started":"2023-01-16T05:41:26.057321Z","shell.execute_reply":"2023-01-16T05:41:26.064106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_classes = np.where(valid_predict>0.5,1,0)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:53:24.957177Z","iopub.execute_input":"2023-01-16T05:53:24.957572Z","iopub.status.idle":"2023-01-16T05:53:24.964755Z","shell.execute_reply.started":"2023-01-16T05:53:24.957538Z","shell.execute_reply":"2023-01-16T05:53:24.963790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_classes","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:53:31.837130Z","iopub.execute_input":"2023-01-16T05:53:31.837538Z","iopub.status.idle":"2023-01-16T05:53:31.844771Z","shell.execute_reply.started":"2023-01-16T05:53:31.837506Z","shell.execute_reply":"2023-01-16T05:53:31.843767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report,confusion_matrix\nconfusion_matrix(y_test,valid_classes)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:54:36.465658Z","iopub.execute_input":"2023-01-16T05:54:36.466051Z","iopub.status.idle":"2023-01-16T05:54:36.515321Z","shell.execute_reply.started":"2023-01-16T05:54:36.466015Z","shell.execute_reply":"2023-01-16T05:54:36.514148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test,valid_classes))","metadata":{"execution":{"iopub.status.busy":"2023-01-16T05:55:20.918649Z","iopub.execute_input":"2023-01-16T05:55:20.919033Z","iopub.status.idle":"2023-01-16T05:55:21.036540Z","shell.execute_reply.started":"2023-01-16T05:55:20.918998Z","shell.execute_reply":"2023-01-16T05:55:21.035290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_dataset[num_columns])\npredictions","metadata":{"execution":{"iopub.status.busy":"2023-01-13T04:05:39.585798Z","iopub.execute_input":"2023-01-13T04:05:39.586166Z","iopub.status.idle":"2023-01-13T04:09:02.030132Z","shell.execute_reply.started":"2023-01-13T04:05:39.586131Z","shell.execute_reply":"2023-01-13T04:09:02.029087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions})\noutput.to_csv('submission_10.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T11:36:49.722980Z","iopub.execute_input":"2023-01-04T11:36:49.723408Z","iopub.status.idle":"2023-01-04T11:36:53.838611Z","shell.execute_reply.started":"2023-01-04T11:36:49.723358Z","shell.execute_reply":"2023-01-04T11:36:53.836855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:37:14.027646Z","iopub.execute_input":"2022-12-22T11:37:14.028191Z","iopub.status.idle":"2022-12-22T11:37:14.038422Z","shell.execute_reply.started":"2022-12-22T11:37:14.028150Z","shell.execute_reply":"2022-12-22T11:37:14.036825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nrf_model = RandomForestClassifier()\nrf_model.fit(x_train, y_train)\npredictions_2= rf_model.predict_proba(test_dataset[num_columns])","metadata":{"execution":{"iopub.status.busy":"2022-12-22T10:58:28.269851Z","iopub.execute_input":"2022-12-22T10:58:28.270282Z","iopub.status.idle":"2022-12-22T11:21:38.248351Z","shell.execute_reply.started":"2022-12-22T10:58:28.270248Z","shell.execute_reply":"2022-12-22T11:21:38.245583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_2[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:22:32.259694Z","iopub.execute_input":"2022-12-22T11:22:32.260113Z","iopub.status.idle":"2022-12-22T11:22:32.269003Z","shell.execute_reply.started":"2022-12-22T11:22:32.260082Z","shell.execute_reply":"2022-12-22T11:22:32.267638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_2[:,1]})\noutput.to_csv('submission_2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:23:09.831466Z","iopub.execute_input":"2022-12-22T11:23:09.832633Z","iopub.status.idle":"2022-12-22T11:23:13.506108Z","shell.execute_reply.started":"2022-12-22T11:23:09.832579Z","shell.execute_reply":"2022-12-22T11:23:13.504739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:25:07.607233Z","iopub.execute_input":"2022-12-22T11:25:07.607759Z","iopub.status.idle":"2022-12-22T11:25:07.623784Z","shell.execute_reply.started":"2022-12-22T11:25:07.607721Z","shell.execute_reply":"2022-12-22T11:25:07.622031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Naive Byes Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.datasets import load_iris\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import GaussianNB\n\ngnb = GaussianNB()\ngnb.fit(x_train, y_train)\npredictions_3= gnb.predict_proba(test_dataset[num_columns])","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:42:43.519264Z","iopub.execute_input":"2022-12-22T11:42:43.519851Z","iopub.status.idle":"2022-12-22T11:42:49.744505Z","shell.execute_reply.started":"2022-12-22T11:42:43.519811Z","shell.execute_reply":"2022-12-22T11:42:49.743195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_3[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:43:44.338285Z","iopub.execute_input":"2022-12-22T11:43:44.338789Z","iopub.status.idle":"2022-12-22T11:43:44.348382Z","shell.execute_reply.started":"2022-12-22T11:43:44.338749Z","shell.execute_reply":"2022-12-22T11:43:44.346889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_3[:, 1]})\noutput.to_csv('submission_3.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:43:25.826178Z","iopub.execute_input":"2022-12-22T11:43:25.826765Z","iopub.status.idle":"2022-12-22T11:43:31.559996Z","shell.execute_reply.started":"2022-12-22T11:43:25.826715Z","shell.execute_reply":"2022-12-22T11:43:31.558629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SVM","metadata":{}},{"cell_type":"code","source":"from sklearn import svm\n\nclf = svm.SVC()\nclf.fit(x_train, y_train)\npredictions_4= clf.predict_proba(test_dataset[num_columns])","metadata":{"execution":{"iopub.status.busy":"2023-01-13T04:13:03.959074Z","iopub.execute_input":"2023-01-13T04:13:03.959516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_4[:, 1]})\noutput.to_csv('submission_4.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Xgboost","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nmodel = XGBClassifier()\nmodel.fit(x_train, y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_5= model.predict_proba(test_dataset[num_columns])","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:57:21.780152Z","iopub.execute_input":"2022-12-22T11:57:21.780588Z","iopub.status.idle":"2022-12-22T11:57:25.583104Z","shell.execute_reply.started":"2022-12-22T11:57:21.780555Z","shell.execute_reply":"2022-12-22T11:57:25.581733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_5[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:58:14.380529Z","iopub.execute_input":"2022-12-22T11:58:14.380989Z","iopub.status.idle":"2022-12-22T11:58:14.390256Z","shell.execute_reply.started":"2022-12-22T11:58:14.380956Z","shell.execute_reply":"2022-12-22T11:58:14.388751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_5[:, 1]})\noutput.to_csv('submission_5.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T11:58:17.909215Z","iopub.execute_input":"2022-12-22T11:58:17.909755Z","iopub.status.idle":"2022-12-22T11:58:22.003277Z","shell.execute_reply.started":"2022-12-22T11:58:17.909716Z","shell.execute_reply":"2022-12-22T11:58:22.002063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adaboost","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier\nadb = AdaBoostClassifier(n_estimators=100, random_state=6)\nadb.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T11:34:15.611582Z","iopub.execute_input":"2022-12-23T11:34:15.612234Z","iopub.status.idle":"2022-12-23T11:51:16.703590Z","shell.execute_reply.started":"2022-12-23T11:34:15.612191Z","shell.execute_reply":"2022-12-23T11:51:16.702637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_6= adb.predict_proba(test_dataset[num_columns])","metadata":{"execution":{"iopub.status.busy":"2022-12-23T11:52:42.488186Z","iopub.execute_input":"2022-12-23T11:52:42.490271Z","iopub.status.idle":"2022-12-23T11:53:35.667034Z","shell.execute_reply.started":"2022-12-23T11:52:42.490195Z","shell.execute_reply":"2022-12-23T11:53:35.665822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_6[:,0]","metadata":{"execution":{"iopub.status.busy":"2022-12-23T11:53:41.112764Z","iopub.execute_input":"2022-12-23T11:53:41.113210Z","iopub.status.idle":"2022-12-23T11:53:41.120968Z","shell.execute_reply.started":"2022-12-23T11:53:41.113173Z","shell.execute_reply":"2022-12-23T11:53:41.120082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_6[:, 1]})\noutput.to_csv('submission_6.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T13:39:16.691911Z","iopub.execute_input":"2022-12-22T13:39:16.692369Z","iopub.status.idle":"2022-12-22T13:39:21.187979Z","shell.execute_reply.started":"2022-12-22T13:39:16.692336Z","shell.execute_reply":"2022-12-22T13:39:21.186853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T13:40:05.286431Z","iopub.execute_input":"2022-12-22T13:40:05.287363Z","iopub.status.idle":"2022-12-22T13:40:05.300278Z","shell.execute_reply.started":"2022-12-22T13:40:05.287312Z","shell.execute_reply":"2022-12-22T13:40:05.298951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KNN","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nneigh = KNeighborsClassifier(n_neighbors=3)\nneigh.fit(x_train, y_train)\npredictions_7= neigh.predict_proba(test_dataset[num_columns])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_7[:, 1]})\noutput.to_csv('submission_7.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Voting Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier","metadata":{"execution":{"iopub.status.busy":"2023-01-04T08:49:31.182229Z","iopub.execute_input":"2023-01-04T08:49:31.182720Z","iopub.status.idle":"2023-01-04T08:49:31.288351Z","shell.execute_reply.started":"2023-01-04T08:49:31.182684Z","shell.execute_reply":"2023-01-04T08:49:31.287204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install catboost\nfrom catboost import CatBoostClassifier","metadata":{"execution":{"iopub.status.busy":"2023-01-04T10:42:06.045659Z","iopub.execute_input":"2023-01-04T10:42:06.046116Z","iopub.status.idle":"2023-01-04T10:42:16.939036Z","shell.execute_reply.started":"2023-01-04T10:42:06.046077Z","shell.execute_reply":"2023-01-04T10:42:16.937770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_1 = CatBoostClassifier()\nmodel_2 = GradientBoostingClassifier()\nmodel_3 = HistGradientBoostingClassifier(max_iter=100)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T08:49:55.356246Z","iopub.execute_input":"2023-01-04T08:49:55.356677Z","iopub.status.idle":"2023-01-04T08:49:55.367165Z","shell.execute_reply.started":"2023-01-04T08:49:55.356641Z","shell.execute_reply":"2023-01-04T08:49:55.365681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier","metadata":{"execution":{"iopub.status.busy":"2023-01-04T08:50:00.382753Z","iopub.execute_input":"2023-01-04T08:50:00.383202Z","iopub.status.idle":"2023-01-04T08:50:00.389399Z","shell.execute_reply.started":"2023-01-04T08:50:00.383170Z","shell.execute_reply":"2023-01-04T08:50:00.387998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_h = VotingClassifier(estimators=[('cb', model_1), ('xgb', model_2), ('gbc', model_3)], voting='soft')","metadata":{"execution":{"iopub.status.busy":"2023-01-04T08:50:15.921318Z","iopub.execute_input":"2023-01-04T08:50:15.921740Z","iopub.status.idle":"2023-01-04T08:50:15.928753Z","shell.execute_reply.started":"2023-01-04T08:50:15.921707Z","shell.execute_reply":"2023-01-04T08:50:15.927284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_h.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T08:50:49.113303Z","iopub.execute_input":"2023-01-04T08:50:49.113812Z","iopub.status.idle":"2023-01-04T09:22:48.814140Z","shell.execute_reply.started":"2023-01-04T08:50:49.113771Z","shell.execute_reply":"2023-01-04T09:22:48.813034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_9 = model_h.predict_proba(test_dataset[num_columns])\npredictions_9","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:23:06.509672Z","iopub.execute_input":"2023-01-04T09:23:06.510185Z","iopub.status.idle":"2023-01-04T09:24:04.151601Z","shell.execute_reply.started":"2023-01-04T09:23:06.510141Z","shell.execute_reply":"2023-01-04T09:24:04.150139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_9[:, 1]})\noutput.to_csv('submission_9.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-04T09:25:35.519340Z","iopub.execute_input":"2023-01-04T09:25:35.519826Z","iopub.status.idle":"2023-01-04T09:25:40.035716Z","shell.execute_reply.started":"2023-01-04T09:25:35.519787Z","shell.execute_reply":"2023-01-04T09:25:40.034248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bagging Classifier","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier","metadata":{"execution":{"iopub.status.busy":"2023-01-04T10:42:32.557108Z","iopub.execute_input":"2023-01-04T10:42:32.557496Z","iopub.status.idle":"2023-01-04T10:42:32.565109Z","shell.execute_reply.started":"2023-01-04T10:42:32.557463Z","shell.execute_reply":"2023-01-04T10:42:32.563606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import BaggingClassifier\nmodel_h = BaggingClassifier(base_estimator=CatBoostClassifier(),n_estimators=100, random_state=0)\n\nmodel_h.fit(x_train, y_train)\n\npredictions_10 = model_h.predict_proba(test_dataset[num_columns])\npredictions_10","metadata":{"execution":{"iopub.status.busy":"2023-01-04T10:42:35.015062Z","iopub.execute_input":"2023-01-04T10:42:35.015560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_dataset = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\noutput = pd.DataFrame({'customer_ID': sample_dataset.customer_ID, 'prediction': predictions_10[:, 1]})\noutput.to_csv('submission_10.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}