{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook provides calculation of XGBoost on a procesed data https://www.kaggle.com/datasets/viktorcikojevic/amex-preprocessed-data.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score\nimport shap\nimport os\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-09T10:42:43.410701Z","iopub.execute_input":"2022-08-09T10:42:43.411041Z","iopub.status.idle":"2022-08-09T10:42:45.562223Z","shell.execute_reply.started":"2022-08-09T10:42:43.411011Z","shell.execute_reply":"2022-08-09T10:42:45.561184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the data\ndf = pd.read_csv(\"../input/amex-preprocessed-data/processed_8_fortraining000000000000.csv\")\nfor i in range (1, 1):\n    print(f\"Loading training {i}/4 ... \")\n    df_tmp = pd.read_csv(f\"../input/amex-preprocessed-data/processed_8_fortraining00000000000{i}.csv\")\n    df = pd.concat([df, df_tmp])\n    del df_tmp","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:11:32.530947Z","iopub.execute_input":"2022-08-09T10:11:32.531398Z","iopub.status.idle":"2022-08-09T10:12:00.901692Z","shell.execute_reply.started":"2022-08-09T10:11:32.531356Z","shell.execute_reply":"2022-08-09T10:12:00.900715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the features and the target\ndf_features = df.drop([\"target\", \"evaluation\", \"D_63_last_value\", \"D_64_last_value\"], axis=1)\ndf_target = df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:12:00.903965Z","iopub.execute_input":"2022-08-09T10:12:00.904372Z","iopub.status.idle":"2022-08-09T10:12:01.085395Z","shell.execute_reply.started":"2022-08-09T10:12:00.904336Z","shell.execute_reply":"2022-08-09T10:12:01.084425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and test sets\nX = df_features\ny = df_target\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.1)\nX_train = X_train.drop([\"customer_ID\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:12:01.087000Z","iopub.execute_input":"2022-08-09T10:12:01.087411Z","iopub.status.idle":"2022-08-09T10:12:01.733041Z","shell.execute_reply.started":"2022-08-09T10:12:01.087373Z","shell.execute_reply":"2022-08-09T10:12:01.731965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the XGBoost\nmodel = xgb.XGBClassifier(tree_method=\"gpu_hist\", \n                          enable_categorical=True, \n                          use_label_encoder=False,\n                          num_parallel_tree=1, \n                          grow_policy=\"lossguide\",\n                         )\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:12:01.735838Z","iopub.execute_input":"2022-08-09T10:12:01.736222Z","iopub.status.idle":"2022-08-09T10:12:18.173332Z","shell.execute_reply.started":"2022-08-09T10:12:01.736168Z","shell.execute_reply":"2022-08-09T10:12:18.172061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a graph\n# xgb.to_graphviz(model, rankdir='LR')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:12:18.174693Z","iopub.execute_input":"2022-08-09T10:12:18.175127Z","iopub.status.idle":"2022-08-09T10:12:18.183813Z","shell.execute_reply.started":"2022-08-09T10:12:18.175082Z","shell.execute_reply":"2022-08-09T10:12:18.182830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_important = model.get_booster().get_score(importance_type='weight')\nkeys = list(feature_important.keys())\nvalues = list(feature_important.values())\n\ndata = pd.DataFrame(data=values, index=keys, columns=[\"score\"]).sort_values(by = \"score\", ascending=False)\ndata.nlargest(40, columns=\"score\").plot(kind='barh', figsize = (20,10)) ## plot top 40 features","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:12:18.187106Z","iopub.execute_input":"2022-08-09T10:12:18.187918Z","iopub.status.idle":"2022-08-09T10:12:18.906544Z","shell.execute_reply.started":"2022-08-09T10:12:18.187875Z","shell.execute_reply":"2022-08-09T10:12:18.905640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make predictions for test data\nX_test_no_customer_ID = X_test.drop(\"customer_ID\", axis=1)\ny_pred = model.predict(X_test_no_customer_ID)\ny_pred_proba = model.predict_proba(X_test_no_customer_ID)\n\n# What is the accuracy of the test data\npredictions = [round(value) for value in y_pred]\n# evaluate predictions\naccuracy = accuracy_score(y_test, predictions)\nprint(\"Accuracy: %.2f%%\" % (accuracy * 100.0))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explainer = shap.TreeExplainer(model)\nshap_values = explainer.shap_values(X_test_no_customer_ID)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:44:26.343180Z","iopub.execute_input":"2022-08-09T10:44:26.344375Z","iopub.status.idle":"2022-08-09T10:44:27.658463Z","shell.execute_reply.started":"2022-08-09T10:44:26.344316Z","shell.execute_reply":"2022-08-09T10:44:27.657470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.summary_plot(shap_values, X_test_no_customer_ID, plot_type=\"bar\", max_display=30)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:57:18.476346Z","iopub.execute_input":"2022-08-09T10:57:18.476812Z","iopub.status.idle":"2022-08-09T10:57:19.881752Z","shell.execute_reply.started":"2022-08-09T10:57:18.476775Z","shell.execute_reply":"2022-08-09T10:57:19.880758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the most wrong predictions\ny_test_list = y_test.to_list()\nprediction_is_true = [prediction == test for prediction, test in zip(y_test, y_pred)]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T10:12:20.606574Z","iopub.execute_input":"2022-08-09T10:12:20.606901Z","iopub.status.idle":"2022-08-09T10:12:20.612886Z","shell.execute_reply.started":"2022-08-09T10:12:20.606874Z","shell.execute_reply":"2022-08-09T10:12:20.611743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# debug\ny_pred_proba_max = np.max(y_pred_proba, axis=1) \nprediction = (y_pred == 1) * y_pred_proba_max  + (y_pred == 0) * (1 - y_pred_proba_max)\nprediction","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:16:13.359739Z","iopub.execute_input":"2022-08-07T17:16:13.360465Z","iopub.status.idle":"2022-08-07T17:16:13.377213Z","shell.execute_reply.started":"2022-08-07T17:16:13.360422Z","shell.execute_reply":"2022-08-07T17:16:13.375602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_sub = pd.DataFrame()\nfor file in os.listdir(\"../input/amex-preprocessed-data\"):\n    if \"test\" not in file:\n        continue\n    print(f\"Loading file {file}...\")\n    df_tmp = pd.read_csv(f\"../input/amex-preprocessed-data/{file}\")\n    customer_id = df_tmp[\"customer_ID\"].to_list()\n    df_features = df_tmp.drop([\"customer_ID\", \"D_63_last_value\", \"D_64_last_value\"], axis=1)\n    print(f\"Making predictions for the file {file}...\")\n    y_pred = model.predict(df_features)\n    y_pred_proba = model.predict_proba(df_features)\n    y_pred_proba_max = np.max(y_pred_proba, axis=1) \n    prediction = (y_pred == 1) * y_pred_proba_max  + (y_pred == 0) * (1 - y_pred_proba_max)\n    del df_tmp\n    df_curr = pd.DataFrame({\"customer_ID\":customer_id,\n                            \"prediction\": prediction})\n    df_sub = pd.concat([df_sub, df_curr])\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:09:11.241315Z","iopub.execute_input":"2022-08-09T11:09:11.242053Z","iopub.status.idle":"2022-08-09T11:09:53.047457Z","shell.execute_reply.started":"2022-08-09T11:09:11.242009Z","shell.execute_reply":"2022-08-09T11:09:53.045930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_sub.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:16:13.380591Z","iopub.execute_input":"2022-08-07T17:16:13.382195Z","iopub.status.idle":"2022-08-07T17:16:13.434618Z","shell.execute_reply.started":"2022-08-07T17:16:13.382150Z","shell.execute_reply":"2022-08-07T17:16:13.433197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-09T11:04:22.903012Z","iopub.execute_input":"2022-08-09T11:04:22.903425Z","iopub.status.idle":"2022-08-09T11:04:22.913827Z","shell.execute_reply.started":"2022-08-09T11:04:22.903388Z","shell.execute_reply":"2022-08-09T11:04:22.912664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}