{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-13T13:20:32.080501Z","iopub.execute_input":"2023-04-13T13:20:32.081083Z","iopub.status.idle":"2023-04-13T13:20:32.088137Z","shell.execute_reply.started":"2023-04-13T13:20:32.081030Z","shell.execute_reply":"2023-04-13T13:20:32.086809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:32.090388Z","iopub.execute_input":"2023-04-13T13:20:32.090839Z","iopub.status.idle":"2023-04-13T13:20:32.101986Z","shell.execute_reply.started":"2023-04-13T13:20:32.090802Z","shell.execute_reply":"2023-04-13T13:20:32.100818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Data\ntrain_data = pd.read_feather('/kaggle/input/amexfeather/train_data.ftr').groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:32.104105Z","iopub.execute_input":"2023-04-13T13:20:32.105102Z","iopub.status.idle":"2023-04-13T13:20:39.456768Z","shell.execute_reply.started":"2023-04-13T13:20:32.105051Z","shell.execute_reply":"2023-04-13T13:20:39.455659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_68']\n\nlabel_encoder = LabelEncoder()\n\nfor feature in categorical_features:\n    train_data[feature] = label_encoder.fit_transform(train_data[feature])","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:39.458796Z","iopub.execute_input":"2023-04-13T13:20:39.459744Z","iopub.status.idle":"2023-04-13T13:20:39.920923Z","shell.execute_reply.started":"2023-04-13T13:20:39.459671Z","shell.execute_reply":"2023-04-13T13:20:39.919522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Data process\n#remove the columns which has lots of missing data\nnull_percentages = (train_data.isnull().sum() / len(train_data)) * 100\nhigh_null_cols = [col for col in null_percentages.index if null_percentages[col] > 80]\ntrain_data = train_data.drop(high_null_cols, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:39.925243Z","iopub.execute_input":"2023-04-13T13:20:39.926212Z","iopub.status.idle":"2023-04-13T13:20:40.663120Z","shell.execute_reply.started":"2023-04-13T13:20:39.926163Z","shell.execute_reply":"2023-04-13T13:20:40.661776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#delete s2 feature\ntrain_data = train_data.drop(['S_2'], axis =1)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:40.664420Z","iopub.execute_input":"2023-04-13T13:20:40.664760Z","iopub.status.idle":"2023-04-13T13:20:40.957836Z","shell.execute_reply.started":"2023-04-13T13:20:40.664729Z","shell.execute_reply":"2023-04-13T13:20:40.956668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categories = [col for col in train_data.columns if train_data[col].dtype == 'category']","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:40.959238Z","iopub.execute_input":"2023-04-13T13:20:40.959586Z","iopub.status.idle":"2023-04-13T13:20:40.973673Z","shell.execute_reply.started":"2023-04-13T13:20:40.959554Z","shell.execute_reply":"2023-04-13T13:20:40.972232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fill null data\nfor i in categories:\n    train_data[i] =  train_data[i].fillna(train_data[i].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:40.975226Z","iopub.execute_input":"2023-04-13T13:20:40.975552Z","iopub.status.idle":"2023-04-13T13:20:40.985243Z","shell.execute_reply.started":"2023-04-13T13:20:40.975522Z","shell.execute_reply":"2023-04-13T13:20:40.983977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_columns = train_data.columns[train_data.isna().any()].tolist()\nfor j in null_columns:\n     train_data[j] = train_data[j].fillna(train_data[j].median())","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:40.987114Z","iopub.execute_input":"2023-04-13T13:20:40.987538Z","iopub.status.idle":"2023-04-13T13:20:42.926745Z","shell.execute_reply.started":"2023-04-13T13:20:40.987501Z","shell.execute_reply":"2023-04-13T13:20:42.925593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nenc = OrdinalEncoder()\ntrain_data[categories] = enc.fit_transform(train_data[categories])","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:42.928542Z","iopub.execute_input":"2023-04-13T13:20:42.928930Z","iopub.status.idle":"2023-04-13T13:20:42.935771Z","shell.execute_reply.started":"2023-04-13T13:20:42.928896Z","shell.execute_reply":"2023-04-13T13:20:42.934640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_data.drop('target', axis=1)\ny = train_data['target']","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:42.937480Z","iopub.execute_input":"2023-04-13T13:20:42.937956Z","iopub.status.idle":"2023-04-13T13:20:43.233089Z","shell.execute_reply.started":"2023-04-13T13:20:42.937920Z","shell.execute_reply":"2023-04-13T13:20:43.231901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_test,y_train,y_test = train_test_split(x, y, test_size=0.22, random_state=42, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:43.234803Z","iopub.execute_input":"2023-04-13T13:20:43.235288Z","iopub.status.idle":"2023-04-13T13:20:44.596398Z","shell.execute_reply.started":"2023-04-13T13:20:43.235254Z","shell.execute_reply":"2023-04-13T13:20:44.595349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Random Forest Classifier\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:44.598080Z","iopub.execute_input":"2023-04-13T13:20:44.598570Z","iopub.status.idle":"2023-04-13T13:20:44.605182Z","shell.execute_reply.started":"2023-04-13T13:20:44.598523Z","shell.execute_reply":"2023-04-13T13:20:44.603911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameter_space = {\n        \"n_estimators\": [20], \n        \"min_samples_leaf\": [31],\n        \"min_samples_split\": [2],\n        \"max_depth\": [10],\n        \"max_features\": [40]\n    }\nclf = RandomForestRegressor(\n        criterion=\"squared_error\",\n        n_jobs=-1,\n        random_state=22)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:44.609828Z","iopub.execute_input":"2023-04-13T13:20:44.610322Z","iopub.status.idle":"2023-04-13T13:20:44.618696Z","shell.execute_reply.started":"2023-04-13T13:20:44.610275Z","shell.execute_reply":"2023-04-13T13:20:44.617414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#grid = GridSearchCV(clf, parameter_space, cv=2, scoring=\"neg_mean_squared_error\")\n#grid.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:44.620389Z","iopub.execute_input":"2023-04-13T13:20:44.620869Z","iopub.status.idle":"2023-04-13T13:20:44.630564Z","shell.execute_reply.started":"2023-04-13T13:20:44.620826Z","shell.execute_reply":"2023-04-13T13:20:44.629407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#XG boost model\nfrom xgboost import XGBClassifier\nmodel = XGBClassifier(n_estimators=300, max_depth=6, learning_rate=0.1).fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:20:44.631878Z","iopub.execute_input":"2023-04-13T13:20:44.632343Z","iopub.status.idle":"2023-04-13T13:33:09.893587Z","shell.execute_reply.started":"2023-04-13T13:20:44.632298Z","shell.execute_reply":"2023-04-13T13:33:09.892292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict= model.predict(x_test)\naccuracy = accuracy_score(y_test, predict)\nprint(\"Accuracy:\", accuracy)\n\n# Evaluate performance\nfrom sklearn.metrics import classification_report\nprint(classification_report(y_test, predict))","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:33:09.895256Z","iopub.execute_input":"2023-04-13T13:33:09.895705Z","iopub.status.idle":"2023-04-13T13:33:10.806952Z","shell.execute_reply.started":"2023-04-13T13:33:09.895660Z","shell.execute_reply":"2023-04-13T13:33:10.805761Z"},"trusted":true},"execution_count":null,"outputs":[]}]}