{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3727003,"sourceType":"datasetVersion","datasetId":2213609}],"dockerImageVersionId":30198,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"* [Introduction](#section-one) \n* [Importing libraries](#section-two) \n* [Loading Data](#section-three)\n* [Checking the size of the data base](#section-four)\n* [Correlation matrix](#section-five)\n\n<a id=\"section-one\"></a>\n# Introduction\n This is my first attempt to create a model to participate in the American Express Challenge to predict credit default.\n \n My main goal is to try to reach a bronze medal, which would require me to rank at least 126th, which has currently a score of 0.796\n \n More on the competition on the link: https://www.kaggle.com/competitions/amex-default-prediction\n \n <a id=\"section-two\"></a>\n# Importing Libraries","metadata":{}},{"cell_type":"code","source":"# Import packages\nimport pandas as pd\nimport numpy as np\nimport pickle\nfrom matplotlib import pyplot as plt\nimport os, gc\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import plot_importance\n\n# Import xgb modules\nimport xgboost as xgb\n\nimport cupy, cudf # GPU libraries","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:37:52.488083Z","iopub.execute_input":"2022-06-17T04:37:52.488691Z","iopub.status.idle":"2022-06-17T04:37:57.668922Z","shell.execute_reply.started":"2022-06-17T04:37:52.488542Z","shell.execute_reply":"2022-06-17T04:37:57.667816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" <a id=\"section-three\"></a>\n# Loading Data","metadata":{}},{"cell_type":"markdown","source":"The code bellow help us open the files in a feather format, which includes all the information, but is much smaller than the original data","metadata":{}},{"cell_type":"code","source":"def read_file_int(path = '', usecols = None):\n    if usecols is not None: df = cudf.read_feather(path, columns=usecols)\n    else: df = cudf.read_feather(path)\n    df.S_2 = cudf.to_datetime(df.S_2)\n    features = [x for x in df.columns.values if x not in ['customer_ID', 'target']]\n    df['n_missing'] = df[features].isna().sum(axis=1)\n    df = df.fillna(NAN_VALUE) \n    df_out = df.groupby(['customer_ID']).nth(-1).reset_index(drop=True)\n    print('shape of data:', df_out.shape)\n    del df\n    _ = gc.collect()\n    return df_out\n\ndef read_file_cpu(path = '', usecols = None):\n    if usecols is not None: df = pd.read_feather(path, columns=usecols)\n    else: df = pd.read_feather(path)\n    df.S_2 = pd.to_datetime(df.S_2)\n    features = [x for x in df.columns.values if x not in ['customer_ID', 'target']]\n    df['n_missing'] = df[features].isna().sum(axis=1)\n    df_out = df.groupby(['customer_ID']).nth(-1).reset_index(drop=True)\n    print('shape of data:', df_out.shape)\n    del df\n    _ = gc.collect()\n    return df_out","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:37:57.671036Z","iopub.execute_input":"2022-06-17T04:37:57.671691Z","iopub.status.idle":"2022-06-17T04:37:57.685618Z","shell.execute_reply.started":"2022-06-17T04:37:57.671635Z","shell.execute_reply":"2022-06-17T04:37:57.684476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading train data...')\nTRAIN_PATH = '../input/amexfeather/train_data.ftr'\ntrain_df = read_file_cpu(path = TRAIN_PATH)\n\nprint('Reading test data...')\nTEST_PATH = '../input/amexfeather/test_data.ftr'\ntest_df = read_file_cpu(path = TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:37:57.691889Z","iopub.execute_input":"2022-06-17T04:37:57.692394Z","iopub.status.idle":"2022-06-17T04:39:45.29194Z","shell.execute_reply.started":"2022-06-17T04:37:57.692354Z","shell.execute_reply":"2022-06-17T04:39:45.291021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" <a id=\"section-three\"></a>\n# Checking the size of the data base","metadata":{}},{"cell_type":"code","source":"from humanize import naturalsize\nsize = train_df.memory_usage(deep='True').sum()\nprint(size)\nprint(naturalsize(size))","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:39:45.295089Z","iopub.execute_input":"2022-06-17T04:39:45.295865Z","iopub.status.idle":"2022-06-17T04:39:45.321889Z","shell.execute_reply.started":"2022-06-17T04:39:45.295829Z","shell.execute_reply":"2022-06-17T04:39:45.321073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train data memory usage: {naturalsize(train_df.memory_usage(deep=\"True\").sum())} ')\nprint(f'Test data memory usage:  {naturalsize(test_df.memory_usage(deep=\"True\").sum())}')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:39:45.323237Z","iopub.execute_input":"2022-06-17T04:39:45.323597Z","iopub.status.idle":"2022-06-17T04:39:45.346448Z","shell.execute_reply.started":"2022-06-17T04:39:45.323562Z","shell.execute_reply":"2022-06-17T04:39:45.345717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" <a id=\"section-five\"></a>\n# Correlation Matrix","metadata":{}},{"cell_type":"markdown","source":"The code bellow finds the most correlated features and exclude them from exclude them from the database.","metadata":{}},{"cell_type":"code","source":"# Correlation matrix\ncorr = train_df.corr()\n# Mask the upper triangle\nmask = np.triu(np.ones_like(corr, dtype=bool))\n# Add the mask to the heatmap\nfig, ax = plt.subplots(1, 1, figsize=(20,14))\nsns.heatmap(corr, mask=mask, center=0, linewidths=1, annot=True, fmt=\".2f\", ax=ax)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:39:45.347766Z","iopub.execute_input":"2022-06-17T04:39:45.348359Z","iopub.status.idle":"2022-06-17T04:41:06.590247Z","shell.execute_reply.started":"2022-06-17T04:39:45.34832Z","shell.execute_reply":"2022-06-17T04:41:06.589327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove highly correlated features\ncorr_matrix = corr.abs()\n# Create a boolean mask and apply it\nmask = np.triu(np.ones_like(corr_matrix, dtype=bool))\ntri_df = corr_matrix.mask(mask)\n\n# List column names of highly correlated features (r > 0.7)\nto_drop = [c for c in tri_df.columns if any(tri_df[c] > 0.7)]\nprint(f'Number of features: {len(to_drop)} \\n {to_drop}')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.591484Z","iopub.execute_input":"2022-06-17T04:41:06.591843Z","iopub.status.idle":"2022-06-17T04:41:06.634351Z","shell.execute_reply.started":"2022-06-17T04:41:06.591809Z","shell.execute_reply":"2022-06-17T04:41:06.633456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code bellow excludes the most hightly correlated features, it's corrently offline.","metadata":{}},{"cell_type":"code","source":"# Remove the highly correlated features\n#train_df = train_df.drop(to_drop, axis=1)\n#test_df = test_df.drop(to_drop, axis=1)\n#train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.635776Z","iopub.execute_input":"2022-06-17T04:41:06.636155Z","iopub.status.idle":"2022-06-17T04:41:06.6413Z","shell.execute_reply.started":"2022-06-17T04:41:06.636117Z","shell.execute_reply":"2022-06-17T04:41:06.63903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Using one hot enconding to transform the categorical collumns","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.642492Z","iopub.execute_input":"2022-06-17T04:41:06.642824Z","iopub.status.idle":"2022-06-17T04:41:06.725982Z","shell.execute_reply.started":"2022-06-17T04:41:06.642785Z","shell.execute_reply":"2022-06-17T04:41:06.725041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.729321Z","iopub.execute_input":"2022-06-17T04:41:06.730052Z","iopub.status.idle":"2022-06-17T04:41:06.874105Z","shell.execute_reply.started":"2022-06-17T04:41:06.730019Z","shell.execute_reply":"2022-06-17T04:41:06.872706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.875805Z","iopub.execute_input":"2022-06-17T04:41:06.876214Z","iopub.status.idle":"2022-06-17T04:41:06.88987Z","shell.execute_reply.started":"2022-06-17T04:41:06.876174Z","shell.execute_reply":"2022-06-17T04:41:06.888784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"According to the chagellenge paga, the categorical collumns are:\n\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{}},{"cell_type":"code","source":"one_hot_encoding_list = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\none_hot_encoding_list_count = len(one_hot_encoding_list)\nprint('There are', one_hot_encoding_list_count, 'variables that will be One Hot Encoded, and they are: \\n', one_hot_encoding_list, '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.891358Z","iopub.execute_input":"2022-06-17T04:41:06.892014Z","iopub.status.idle":"2022-06-17T04:41:06.897863Z","shell.execute_reply.started":"2022-06-17T04:41:06.891975Z","shell.execute_reply":"2022-06-17T04:41:06.897048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# Apply one-hot encoder to each column with categorical data\nOH_encoder = OneHotEncoder(handle_unknown='ignore', sparse=False)\n\none_hot_encoding_train_df = pd.DataFrame(OH_encoder.fit_transform(train_df[one_hot_encoding_list]))\none_hot_encoding_test_df = pd.DataFrame(OH_encoder.transform(test_df[one_hot_encoding_list]))\n\n# One-hot encoding removed index; put it back\none_hot_encoding_train_df.index = train_df.index\none_hot_encoding_test_df.index = test_df.index\n\none_hot_encoding_train_df.columns = OH_encoder.get_feature_names_out()\none_hot_encoding_test_df.columns = one_hot_encoding_train_df.columns\n\none_hot_encoding_train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:06.899201Z","iopub.execute_input":"2022-06-17T04:41:06.900247Z","iopub.status.idle":"2022-06-17T04:41:09.783252Z","shell.execute_reply.started":"2022-06-17T04:41:06.900211Z","shell.execute_reply":"2022-06-17T04:41:09.782428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dates = train_df['S_2']\ndates_test = test_df['S_2']\ndates","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:09.784634Z","iopub.execute_input":"2022-06-17T04:41:09.785242Z","iopub.status.idle":"2022-06-17T04:41:09.795296Z","shell.execute_reply.started":"2022-06-17T04:41:09.785199Z","shell.execute_reply":"2022-06-17T04:41:09.794281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_features = train_df._get_numeric_data()\nnum_features_test = test_df._get_numeric_data()\nnum_features","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:09.796666Z","iopub.execute_input":"2022-06-17T04:41:09.797312Z","iopub.status.idle":"2022-06-17T04:41:09.880794Z","shell.execute_reply.started":"2022-06-17T04:41:09.797271Z","shell.execute_reply":"2022-06-17T04:41:09.879838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.concat([dates, num_features, one_hot_encoding_train_df], axis = 1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:09.882166Z","iopub.execute_input":"2022-06-17T04:41:09.882631Z","iopub.status.idle":"2022-06-17T04:41:10.434683Z","shell.execute_reply.started":"2022-06-17T04:41:09.882582Z","shell.execute_reply":"2022-06-17T04:41:10.433685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['S_2']","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:10.436167Z","iopub.execute_input":"2022-06-17T04:41:10.436683Z","iopub.status.idle":"2022-06-17T04:41:10.445697Z","shell.execute_reply.started":"2022-06-17T04:41:10.436626Z","shell.execute_reply":"2022-06-17T04:41:10.444699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.concat([dates_test, num_features_test, one_hot_encoding_test_df], axis = 1)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:10.447563Z","iopub.execute_input":"2022-06-17T04:41:10.448031Z","iopub.status.idle":"2022-06-17T04:41:11.641474Z","shell.execute_reply.started":"2022-06-17T04:41:10.447913Z","shell.execute_reply":"2022-06-17T04:41:11.640532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.fillna(0)\ntest_df = test_df.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:11.643112Z","iopub.execute_input":"2022-06-17T04:41:11.643501Z","iopub.status.idle":"2022-06-17T04:41:23.772175Z","shell.execute_reply.started":"2022-06-17T04:41:11.643463Z","shell.execute_reply":"2022-06-17T04:41:23.771295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:23.773692Z","iopub.execute_input":"2022-06-17T04:41:23.774087Z","iopub.status.idle":"2022-06-17T04:41:23.869949Z","shell.execute_reply.started":"2022-06-17T04:41:23.774048Z","shell.execute_reply":"2022-06-17T04:41:23.868989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:23.871471Z","iopub.execute_input":"2022-06-17T04:41:23.871872Z","iopub.status.idle":"2022-06-17T04:41:24.04711Z","shell.execute_reply.started":"2022-06-17T04:41:23.871834Z","shell.execute_reply":"2022-06-17T04:41:24.046154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['day_of_month'] = train_df['S_2'].dt.day\ntrain_df['month'] = train_df['S_2'].dt.month\ntrain_df['year'] = train_df['S_2'].dt.year\n\ntest_df['day_of_month'] = test_df['S_2'].dt.day\ntest_df['month'] = test_df['S_2'].dt.month\ntest_df['year'] = test_df['S_2'].dt.year","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:24.048716Z","iopub.execute_input":"2022-06-17T04:41:24.049364Z","iopub.status.idle":"2022-06-17T04:41:24.407587Z","shell.execute_reply.started":"2022-06-17T04:41:24.049318Z","shell.execute_reply":"2022-06-17T04:41:24.406671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:24.409155Z","iopub.execute_input":"2022-06-17T04:41:24.409562Z","iopub.status.idle":"2022-06-17T04:41:24.516346Z","shell.execute_reply.started":"2022-06-17T04:41:24.409522Z","shell.execute_reply":"2022-06-17T04:41:24.515508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:24.517772Z","iopub.execute_input":"2022-06-17T04:41:24.518742Z","iopub.status.idle":"2022-06-17T04:41:24.708112Z","shell.execute_reply.started":"2022-06-17T04:41:24.5187Z","shell.execute_reply":"2022-06-17T04:41:24.707086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"D_* = Delinquency variables\nS_* = Spend variables\nP_* = Payment variables\nB_* = Balance variables\nR_* = Risk variables\n\ndelinquency_variables = ['D_39', 'D_41', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_51', 'D_52', 'D_53', 'D_54', 'D_55', 'D_56', 'D_58', 'D_59', \n                         'D_60', 'D_61', 'D_62', 'D_65', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_74', 'D_75', 'D_76', 'D_77', 'D_78', 'D_79', 'D_80', 'D_81', 'D_82', 'D_83', \n                         'D_84', 'D_86', 'D_87', 'D_88', 'D_89', 'D_91', 'D_92', 'D_93', 'D_94', 'D_96', 'D_102', 'D_103', 'D_104', 'D_105', 'D_106', 'D_107', 'D_108', 'D_109', \n                         'D_110', 'D_111', 'D_112', 'D_113', 'D_115', 'D_118', 'D_119', 'D_121', 'D_122', 'D_123', 'D_124', 'D_125', 'D_127', 'D_128', 'D_129', 'D_130', 'D_131',\n                         'D_132', 'D_133', 'D_134', 'D_135', 'D_136', 'D_137', 'D_138', 'D_139', 'D_140', 'D_141', 'D_142', 'D_143', 'D_144', 'D_145']\n\n\nspend_variables = ['S_3', 'S_5', 'S_6', 'S_7', 'S_8', 'S_9', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_18', 'S_19', 'S_20', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\n\npayment_variables = ['P_2', 'P_3', 'P_4']\n\nbalance_variables = ['B_1', 'B_2', 'B_3', 'B_4','B_6', 'B_7', 'B_8','B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15','B_16','B_17','B_18','B_19','B_20','B_21','B_22',\n                     'B_23','B_24', 'B_25', 'B_26','B_27','B_28','B_29', 'B_31', 'B_32', 'B_33', 'B_36', 'B_37', 'B_40', 'B_41', 'B_42']\n\nrisk_variables = ['R_1', 'R_2', 'R_3', 'R_4', 'R_5', 'R_6', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_12', 'R_13', 'R_14', 'R_15', 'R_16', 'R_17', 'R_18', 'R_19',\n                  'R_20', 'R_21', 'R_22', 'R_23', 'R_24', 'R_25', 'R_26', 'R_27', 'R_28']\n\nD = train_df[delinquency_variables]\nS = train_df[spend_variables]\nP = train_df[payment_variables]\nB = train_df[balance_variables]\nR = train_df[risk_variables]\n\nD_test = test_df[delinquency_variables]\nS_test = test_df[spend_variables]\nP_test = test_df[payment_variables]\nB_test = test_df[balance_variables]\nR_test = test_df[risk_variables]\n\nfrom sklearn.cluster import KMeans\nkmeans = KMeans(n_clusters=4)\ntrain_df[\"Cluster1\"] = kmeans.fit_predict(D)\ntrain_df[\"Cluster2\"] = kmeans.fit_predict(S)\ntrain_df[\"Cluster3\"] = kmeans.fit_predict(P)\ntrain_df[\"Cluster4\"] = kmeans.fit_predict(B)\ntrain_df[\"Cluster5\"] = kmeans.fit_predict(R)\n\ntest_df[\"Cluster1\"] = kmeans.fit_predict(D_test)\ntest_df[\"Cluster2\"] = kmeans.fit_predict(S_test)\ntest_df[\"Cluster3\"] = kmeans.fit_predict(P_test)\ntest_df[\"Cluster4\"] = kmeans.fit_predict(B_test)\ntest_df[\"Cluster5\"] = kmeans.fit_predict(R_test)\n\ntrain_df","metadata":{}},{"cell_type":"code","source":"variables = ['D_39', 'P_2']\n\nd_59 = train_df[variables]\nd_59_test = test_df[variables]\n\nfrom sklearn.cluster import KMeans \nkmeans = KMeans(n_clusters=4) \ntrain_df[\"Cluster\"] = kmeans.fit_predict(d_59) \ntest_df[\"Cluster\"] = kmeans.fit_predict(d_59_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:24.709561Z","iopub.execute_input":"2022-06-17T04:41:24.710074Z","iopub.status.idle":"2022-06-17T04:41:31.368077Z","shell.execute_reply.started":"2022-06-17T04:41:24.710031Z","shell.execute_reply":"2022-06-17T04:41:31.36706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data set for analysis\n\nThe code bellow separates Target from Train","metadata":{}},{"cell_type":"code","source":"target = train_df['target']\ntrain_df = train_df.drop(['target'], axis=1)\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:31.369599Z","iopub.execute_input":"2022-06-17T04:41:31.369996Z","iopub.status.idle":"2022-06-17T04:41:31.701384Z","shell.execute_reply.started":"2022-06-17T04:41:31.369959Z","shell.execute_reply":"2022-06-17T04:41:31.700555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competition Metrics\n\nThe competition is metric performance is measured according to the code bellow","metadata":{}},{"cell_type":"code","source":"def amex_metric_numpy(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = np.sum(y_true)\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by describing prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = np.sum(target[four_pct_mask]) / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:31.702895Z","iopub.execute_input":"2022-06-17T04:41:31.703376Z","iopub.status.idle":"2022-06-17T04:41:31.713079Z","shell.execute_reply.started":"2022-06-17T04:41:31.703335Z","shell.execute_reply":"2022-06-17T04:41:31.711026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"num_features = train_df._get_numeric_data().columns\nnum_features","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:31.718114Z","iopub.execute_input":"2022-06-17T04:41:31.718439Z","iopub.status.idle":"2022-06-17T04:41:31.725398Z","shell.execute_reply.started":"2022-06-17T04:41:31.718413Z","shell.execute_reply":"2022-06-17T04:41:31.724612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the arrays for features and the target: X, y\nX, y = train_df._get_numeric_data(), target","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:31.726961Z","iopub.execute_input":"2022-06-17T04:41:31.727759Z","iopub.status.idle":"2022-06-17T04:41:31.735334Z","shell.execute_reply.started":"2022-06-17T04:41:31.727695Z","shell.execute_reply":"2022-06-17T04:41:31.734546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the training and test datasets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state=100, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:31.736761Z","iopub.execute_input":"2022-06-17T04:41:31.737161Z","iopub.status.idle":"2022-06-17T04:41:33.048578Z","shell.execute_reply.started":"2022-06-17T04:41:31.737101Z","shell.execute_reply":"2022-06-17T04:41:33.047704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_cl = XGBClassifier(objective='binary:logistic', \n                      n_estimators=100,\n                      seed=123,\n                      use_label_encoder=False,\n                      eval_metric='aucpr', # updated to make use of the aucpr option\n                      early_stopping_rounds=10,\n                      tree_method='gpu_hist',\n                      enable_categorical=True\n                      )\neval_set = [(X_test, y_test)]","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:33.050133Z","iopub.execute_input":"2022-06-17T04:41:33.050535Z","iopub.status.idle":"2022-06-17T04:41:33.056976Z","shell.execute_reply.started":"2022-06-17T04:41:33.050497Z","shell.execute_reply":"2022-06-17T04:41:33.055861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the classifier\nxg_cl.fit(X_train, y_train, eval_set=eval_set, verbose=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:33.058451Z","iopub.execute_input":"2022-06-17T04:41:33.058886Z","iopub.status.idle":"2022-06-17T04:41:39.772445Z","shell.execute_reply.started":"2022-06-17T04:41:33.05885Z","shell.execute_reply":"2022-06-17T04:41:39.771573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the labels of the test set\npreds = xg_cl.predict(X_test)\npreds_prob = xg_cl.predict_proba(X_test)[:,1]\n\n# Compute accuracy\naccuracy = accuracy_score(y_test, preds)\nprint(f'accuracy: {accuracy: .2%}')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:39.773704Z","iopub.execute_input":"2022-06-17T04:41:39.77408Z","iopub.status.idle":"2022-06-17T04:41:40.603035Z","shell.execute_reply.started":"2022-06-17T04:41:39.774041Z","shell.execute_reply":"2022-06-17T04:41:40.602237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Review the important features\n# print(xg_cl.feature_importances_)\ndef plot_features(booster, figsize, max_num_features=15):    \n    fig, ax = plt.subplots(1,1,figsize=figsize)\n    return plot_importance(booster=booster, ax=ax, max_num_features=max_num_features)\nplot_features(xg_cl, (10,14))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:40.605143Z","iopub.execute_input":"2022-06-17T04:41:40.606087Z","iopub.status.idle":"2022-06-17T04:41:40.879391Z","shell.execute_reply.started":"2022-06-17T04:41:40.606036Z","shell.execute_reply":"2022-06-17T04:41:40.878539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Metric Evaluation Values\\n')\nprint(f'Numpy: {amex_metric_numpy(y_test.to_numpy().ravel(), preds_prob)}')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:40.880923Z","iopub.execute_input":"2022-06-17T04:41:40.881323Z","iopub.status.idle":"2022-06-17T04:41:40.898298Z","shell.execute_reply.started":"2022-06-17T04:41:40.881286Z","shell.execute_reply":"2022-06-17T04:41:40.897537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understanding weighted class imbalance\nfrom collections import Counter\n\ncounter = Counter(y)\nprint(counter)\n\n# estimate scale_pos_weight value\nestimate = counter[0] / counter[1]\nprint('Estimate: %.3f' % estimate)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:40.899836Z","iopub.execute_input":"2022-06-17T04:41:40.900615Z","iopub.status.idle":"2022-06-17T04:41:40.979115Z","shell.execute_reply.started":"2022-06-17T04:41:40.900557Z","shell.execute_reply":"2022-06-17T04:41:40.977927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the DMatrix from X and y: churn_dmatrix\nd_train = xgb.DMatrix(data=X_train, label=y_train)\nd_test = xgb.DMatrix(data=X_test, label=y_test)\nxgd_test = xgb.DMatrix(data=test_df._get_numeric_data())\n\n# Create the parameter dictionary: params. NOTE: have to explicitly provide the objective param\nparams = {\"objective\":\"binary:logistic\", \n          \"max_depth\": 6,\n          \"eval_metric\":'aucpr', # updated to make use of the aucpr option\n          \"tree_method\":'gpu_hist',\n          \"predictor\": 'gpu_predictor',\n#           \"scale_pos_weight\": 30,\n         }\n\n# Reviewing the AUC metric\n# Perform cross_validation: cv_results\ncv_results = xgb.cv(dtrain=d_train, params=params,\n                    nfold=5, num_boost_round=60, \n                    metrics=\"aucpr\", as_pandas=True, seed=123)\n\n# Print cv_results\nprint(cv_results)\n\n# Print the AUC\n# print((cv_results[\"test-auc-mean\"]).iloc[-1])\nprint((cv_results[\"test-aucpr-mean\"]).iloc[-1])","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:41:40.98063Z","iopub.execute_input":"2022-06-17T04:41:40.981189Z","iopub.status.idle":"2022-06-17T04:42:09.724524Z","shell.execute_reply.started":"2022-06-17T04:41:40.981151Z","shell.execute_reply":"2022-06-17T04:42:09.722677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Review the train method\nparams = {\n    \"objective\":\"binary:logistic\", \n    \"max_depth\": 6,\n    \"eval_metric\":'aucpr', \n    \"tree_method\":'gpu_hist',\n    \"predictor\": 'gpu_predictor',\n#     \"scale_pos_weight\": 30,    \n}\n\n# train - verbose_eval option switches off the log outputs\nxgb_clf = xgb.train(\n    params,\n    d_train,\n    num_boost_round=5000,\n    evals=[(d_train, 'train'), (d_test, 'test')],\n    early_stopping_rounds=10,\n    verbose_eval=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:09.725872Z","iopub.execute_input":"2022-06-17T04:42:09.726345Z","iopub.status.idle":"2022-06-17T04:42:11.974596Z","shell.execute_reply.started":"2022-06-17T04:42:09.726303Z","shell.execute_reply":"2022-06-17T04:42:11.973967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict\ny_pred = xgb_clf.predict(d_test)\n\n# Compute and print metrics\nprint('Metric Evaluation Values\\n')\nprint(f'Numpy: {amex_metric_numpy(y_test.to_numpy().ravel(), y_pred)}')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:11.977805Z","iopub.execute_input":"2022-06-17T04:42:11.978497Z","iopub.status.idle":"2022-06-17T04:42:12.007448Z","shell.execute_reply.started":"2022-06-17T04:42:11.978454Z","shell.execute_reply":"2022-06-17T04:42:12.006365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df._get_numeric_data().columns","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:12.008858Z","iopub.execute_input":"2022-06-17T04:42:12.009251Z","iopub.status.idle":"2022-06-17T04:42:12.016892Z","shell.execute_reply.started":"2022-06-17T04:42:12.009212Z","shell.execute_reply":"2022-06-17T04:42:12.016014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets build using the X_test data - this was to check and see if the code worked. Now going to score up the train_df to get a larger sample\n# 1. Predict probability\n# rank_data = X_test.copy()\n# rank_data['target'] = y_test\n# rank_data['prob'] = preds_prob\n# rank_data.head()\n\nrank_data = train_df._get_numeric_data()\nxgd_rank = xgb.DMatrix(data=train_df._get_numeric_data())\nrank_data['prob'] = xgb_clf.predict(xgd_rank)\nrank_data['target'] = target\nrank_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:12.018415Z","iopub.execute_input":"2022-06-17T04:42:12.018868Z","iopub.status.idle":"2022-06-17T04:42:15.718516Z","shell.execute_reply.started":"2022-06-17T04:42:12.018832Z","shell.execute_reply":"2022-06-17T04:42:15.717611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First create the decile value by prob\nrank = rank_data.loc[:, ['target', 'prob']]\nrank[\"ranks\"] = rank['prob'].rank(method=\"first\")\n\n# The notes displayed here had related to only using the X_test dataframe. With the train_df being used we can try using the probabilities again\n# First method bunchs the final three buckets into one as there are a low of low probs\nrank['decile'] = pd.qcut(rank.prob, 10, labels=False, duplicates='drop') \n# Second method aims to use the rank method, however the nature of this rank is still random\n# An alternative for this piece might be to put the 'prob' in order and sort by the target\n# rank['decile'] = pd.qcut(rank.ranks, 10, labels=False)\n# Reviewing the lowest probability\nmin_prob = np.min(rank.prob)\nrank.loc[(rank.prob == min_prob)].head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:15.720012Z","iopub.execute_input":"2022-06-17T04:42:15.720412Z","iopub.status.idle":"2022-06-17T04:42:15.88004Z","shell.execute_reply.started":"2022-06-17T04:42:15.720371Z","shell.execute_reply":"2022-06-17T04:42:15.879099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a rank_order table\ndef rank_order(df: pd.DataFrame, y: str, target: str) -> pd.DataFrame:\n    \n    rank = df.groupby('decile').apply(lambda x: pd.Series([\n        np.min(x[y]),\n        np.max(x[y]),\n        np.mean(x[y]),\n        np.size(x[y]),\n        np.sum(x[target]),\n        np.size(x[target][x[target]==0]),\n    ],\n        index=([\"min_prob\",\"max_prob\",\"avg_prob\",\n               \"cnt_cust\",\"cnt_def\",\"cnt_non_def\"])\n    )).reset_index()\n    rank = rank.sort_values(by='decile', ascending=False)\n    rank[\"drate\"] = round(rank[\"cnt_def\"]*100/rank[\"cnt_cust\"], 2)\n    rank[\"cum_cust\"] = np.cumsum(rank[\"cnt_cust\"])\n    rank[\"cum_def\"] = np.cumsum(rank[\"cnt_def\"])\n    rank[\"cum_non_def\"] = np.cumsum(rank[\"cnt_non_def\"])\n    rank[\"cum_cust_pct\"] = round(rank[\"cum_cust\"]*100/np.sum(rank[\"cnt_cust\"]), 2)\n    rank[\"cum_def_pct\"] = round(rank[\"cum_def\"]*100/np.sum(rank[\"cnt_def\"]), 2)\n    rank[\"cum_non_def_pct\"] = round(rank[\"cum_non_def\"]*100/np.sum(rank[\"cnt_non_def\"]), 2)\n    rank[\"KS\"] = round(rank[\"cum_def_pct\"] - rank[\"cum_non_def_pct\"],2)\n    rank[\"Lift\"] = round(rank[\"cum_def_pct\"] / rank[\"cum_non_def_pct\"],2)\n    return rank\n\nrank_gains_table = rank_order(rank, \"prob\", \"target\")\nrank_gains_table","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:15.881718Z","iopub.execute_input":"2022-06-17T04:42:15.882163Z","iopub.status.idle":"2022-06-17T04:42:15.968674Z","shell.execute_reply.started":"2022-06-17T04:42:15.88212Z","shell.execute_reply":"2022-06-17T04:42:15.967715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Score up the test dataset\ntest_preds = xgb_clf.predict(xgd_test)\ntest_preds.view()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:15.970185Z","iopub.execute_input":"2022-06-17T04:42:15.970576Z","iopub.status.idle":"2022-06-17T04:42:16.454442Z","shell.execute_reply.started":"2022-06-17T04:42:15.970537Z","shell.execute_reply":"2022-06-17T04:42:16.453546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make submission\nsub_data = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsub_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:16.455933Z","iopub.execute_input":"2022-06-17T04:42:16.456332Z","iopub.status.idle":"2022-06-17T04:42:17.947964Z","shell.execute_reply.started":"2022-06-17T04:42:16.456294Z","shell.execute_reply":"2022-06-17T04:42:17.947157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_data['prediction'] = test_preds\nsub_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:17.94918Z","iopub.execute_input":"2022-06-17T04:42:17.950029Z","iopub.status.idle":"2022-06-17T04:42:17.961748Z","shell.execute_reply.started":"2022-06-17T04:42:17.949987Z","shell.execute_reply":"2022-06-17T04:42:17.960752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission file\nsub_data.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:17.963196Z","iopub.execute_input":"2022-06-17T04:42:17.963984Z","iopub.status.idle":"2022-06-17T04:42:22.915137Z","shell.execute_reply.started":"2022-06-17T04:42:17.963944Z","shell.execute_reply":"2022-06-17T04:42:22.914013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T04:42:22.920032Z","iopub.execute_input":"2022-06-17T04:42:22.920739Z","iopub.status.idle":"2022-06-17T04:42:22.971023Z","shell.execute_reply.started":"2022-06-17T04:42:22.920709Z","shell.execute_reply":"2022-06-17T04:42:22.970058Z"},"trusted":true},"execution_count":null,"outputs":[]}]}