{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-19T18:18:28.280221Z","iopub.execute_input":"2022-08-19T18:18:28.280898Z","iopub.status.idle":"2022-08-19T18:18:28.295060Z","shell.execute_reply.started":"2022-08-19T18:18:28.280853Z","shell.execute_reply":"2022-08-19T18:18:28.293902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline\n\n1. Based a lot of work here off the top solutions. Also thank you https://www.kaggle.com/datasets/munumbutt/amexfeather for the feather dataset\n2. Approach - loaded data, converted to numerics, train-test split, xgb\n3. Building on soln 1 by trying a few more things like - permutation importance\n","metadata":{}},{"cell_type":"code","source":"# Load Data\nimport pandas as pd, numpy as np # CPU libraries\nimport gc\n\n\ntrain = pd.read_feather('../input/amexfeather/train_data.ftr')\nprint(train.shape)\n#Reduce Train (only reduced for testing)\npartTrain = train.iloc[0:5531451, :]\nprint(partTrain.shape)\n\ndel train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:18:28.297521Z","iopub.execute_input":"2022-08-19T18:18:28.298135Z","iopub.status.idle":"2022-08-19T18:18:52.574561Z","shell.execute_reply.started":"2022-08-19T18:18:28.298098Z","shell.execute_reply":"2022-08-19T18:18:52.573121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#List non-numeric columns\nallCols = partTrain.columns.tolist()\nnumericCols = partTrain.select_dtypes(include=np.number).columns.tolist()\n\nprint(\"all\", len(allCols))\nprint(\"numeric\", len(numericCols))\n\nnonNumeric = list(set(allCols) - set(numericCols))\nprint(\"non numeric\", len(nonNumeric))\n# converting pandas \"categorical\" dtype to numeric\npartTrain[nonNumeric] = partTrain[nonNumeric].apply(pd.to_numeric, errors='coerce')\n\n\ndel allCols, numericCols\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:18:52.576471Z","iopub.execute_input":"2022-08-19T18:18:52.580772Z","iopub.status.idle":"2022-08-19T18:18:52.933529Z","shell.execute_reply.started":"2022-08-19T18:18:52.580720Z","shell.execute_reply":"2022-08-19T18:18:52.931991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_test split\nX = partTrain.drop(columns=[\"target\"],axis=1)\ny = partTrain[\"target\"]\n\ndel partTrain\n \nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.33,random_state=100)\n\ndel X\ndel y\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:18:52.943217Z","iopub.execute_input":"2022-08-19T18:18:52.943771Z","iopub.status.idle":"2022-08-19T18:18:53.256122Z","shell.execute_reply.started":"2022-08-19T18:18:52.943722Z","shell.execute_reply":"2022-08-19T18:18:53.254929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#implement XGB baseline\n\nfrom xgboost import XGBClassifier\n\n# Create as many classifiers as there are clusters\nxgb_classifier0 = XGBClassifier(objective='binary:logistic', \n                      n_estimators=200,\n                      eta=0.2,\n                      seed=12,\n                      learning_rate=0.02,\n                      use_label_encoder=False,\n                      eval_metric='aucpr',                      \n                    )\n\n\nprint(\"Print Xtrain\", X_train.shape)\nprint(\"Print y_train\", y_train.shape)\nxgb_classifier0.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:18:53.257960Z","iopub.execute_input":"2022-08-19T18:18:53.258814Z","iopub.status.idle":"2022-08-19T18:19:06.385152Z","shell.execute_reply.started":"2022-08-19T18:18:53.258747Z","shell.execute_reply":"2022-08-19T18:19:06.383850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/anjaroed/permutation-importance-and-xgbclassifier\n\n#Permutation importance\nfrom sklearn.inspection import permutation_importance\nimport matplotlib.pyplot as plt\n\nresult = permutation_importance(xgb_classifier0, X_test, y_test, n_repeats=10, random_state=42, n_jobs=2)\nsorted_idx = result.importances_mean.argsort()\n\nfig, ax = plt.subplots()\nax.boxplot(result.importances[sorted_idx].T,\n           vert=False, labels=X_test.columns[sorted_idx])\nax.set_title(\"Permutation Importances (test set)\")\nfig.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:19:06.386497Z","iopub.execute_input":"2022-08-19T18:19:06.387225Z","iopub.status.idle":"2022-08-19T18:19:42.066880Z","shell.execute_reply.started":"2022-08-19T18:19:06.387189Z","shell.execute_reply":"2022-08-19T18:19:42.065134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Drop Columns that have low imporance\nprint(type(result.importances_mean))\nprint(len(result.importances_mean))\nimportanceArr = result.importances_mean\ncolumnArr = np.array(X_test.columns.tolist())\npermutationImportanceDropList = np.take(columnArr, np.where(importanceArr < 0))\nprint(permutationImportanceDropList[0])\nprint(X_train.shape)\nX_train = X_train.drop(permutationImportanceDropList[0], axis=1)\nprint(X_train.shape)  \n\n\n#test1Arr = np.array(['hello','goodbye','yes'])\n#test2Arr = np.array([1,-1,-5])\n#print(np.where(test2Arr < 0))\n#print(np.take(test1Arr, np.where(test2Arr < 0)))\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:19:42.068994Z","iopub.execute_input":"2022-08-19T18:19:42.069450Z","iopub.status.idle":"2022-08-19T18:19:42.084842Z","shell.execute_reply.started":"2022-08-19T18:19:42.069407Z","shell.execute_reply":"2022-08-19T18:19:42.083699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop Columns that are highly correlated\n# Referecnce - https://www.kaggle.com/code/kellibelcher/amex-default-prediction-eda-lgbm-baseline\nimport seaborn as sns\n\n# Create correlation matrix\ncorr_matrix = X_train.corr().abs()\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(np.bool))\n\n# Find features with correlation greater than 0.95\ncorrDropList = [column for column in upper.columns if any(upper[column] > 0.99)]\nprint(\"To Drop\", corrDropList)\n\n#View Correlation example\nprint(corrDropList[0])\nsns.pairplot(X_train[[corrDropList[0], corrDropList[1]]])\nsns.pairplot(X_train[[corrDropList[1], corrDropList[2]]])\n\n\n\nprint(X_train.shape)\nX_train = X_train.drop(corrDropList, axis=1)\nprint(X_train.shape)  ","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:19:42.086932Z","iopub.execute_input":"2022-08-19T18:19:42.087666Z","iopub.status.idle":"2022-08-19T18:19:44.661084Z","shell.execute_reply.started":"2022-08-19T18:19:42.087628Z","shell.execute_reply":"2022-08-19T18:19:44.659613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Re-train classifier\n\nxgb_classifier1 = XGBClassifier(objective='binary:logistic', \n                      n_estimators=200,\n                      eta=0.2,\n                      seed=12,\n                      learning_rate=0.02,\n                      use_label_encoder=False,\n                      eval_metric='aucpr',                      \n                    )\nprint(\"XGB train\")\nxgb_classifier1.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:19:44.662721Z","iopub.execute_input":"2022-08-19T18:19:44.663151Z","iopub.status.idle":"2022-08-19T18:19:55.632976Z","shell.execute_reply.started":"2022-08-19T18:19:44.663114Z","shell.execute_reply":"2022-08-19T18:19:55.631611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LGBM model\n# ref : https://www.kaggle.com/code/ambrosm/amex-lightgbm-quickstart\nimport lightgbm as lgb\n\nlgbmClassifier = lgb.LGBMClassifier(n_estimators=100,\n                        learning_rate=0.03, \n                        reg_lambda=50,\n                        min_child_samples=2400,\n                        num_leaves=95,\n                        colsample_bytree=0.19,\n                        max_bins=511, random_state=5)\n\n\nX_test = X_test.drop(permutationImportanceDropList[0], axis=1)\nX_test = X_test.drop(corrDropList, axis=1)\n\nprint(\"LGBM train\")\nlgbmClassifier.fit(X_train, y_train, eval_set = [(X_test, y_test)])","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:19:55.636785Z","iopub.execute_input":"2022-08-19T18:19:55.637311Z","iopub.status.idle":"2022-08-19T18:19:55.713179Z","shell.execute_reply.started":"2022-08-19T18:19:55.637273Z","shell.execute_reply":"2022-08-19T18:19:55.712205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#generate output\n\ngc.collect()\n\nimport pandas as pd # Remove when running from the beginning\npartTest = pd.read_feather('../input/amexfeather/test_data.ftr')\npartTest = partTest.drop_duplicates(['customer_ID'])\n\npartTest[nonNumeric] = partTest[nonNumeric].apply(pd.to_numeric, errors='coerce')\npartTest = partTest.drop(permutationImportanceDropList[0], axis=1)\npartTest = partTest.drop(corrDropList, axis=1)\n\nprint(\"XGB prediction\")\ny_pred_prob_xgb = xgb_classifier1.predict_proba(partTest)[:,1]\nprint(y_pred_prob_xgb.shape)\n\nprint(\"LGBM prediction\")\ny_pred_prob_lgbm = lgbmClassifier.predict_proba(partTest)[:,1]\nprint(y_pred_prob_lgbm.shape)\n\ny_pred_prob = (y_pred_prob_xgb + y_pred_prob_lgbm)/2","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:19:55.717600Z","iopub.execute_input":"2022-08-19T18:19:55.719059Z","iopub.status.idle":"2022-08-19T18:20:21.318488Z","shell.execute_reply.started":"2022-08-19T18:19:55.718991Z","shell.execute_reply":"2022-08-19T18:20:21.317493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Submission\nsubmission = pd.read_feather('../input/amexfeather/test_data.ftr')\nsubmission = submission.drop_duplicates(['customer_ID'])\nsubmission[\"prediction\"] = y_pred_prob\nsubmission = submission[['customer_ID', \"prediction\"]]\nprint(submission.head, submission.shape)\nsubmission.to_csv('submission.csv',index=False)\nprint('Submission file shape is', submission.shape )","metadata":{"execution":{"iopub.status.busy":"2022-08-19T18:20:21.319955Z","iopub.execute_input":"2022-08-19T18:20:21.320586Z","iopub.status.idle":"2022-08-19T18:20:37.983573Z","shell.execute_reply.started":"2022-08-19T18:20:21.320543Z","shell.execute_reply":"2022-08-19T18:20:37.982236Z"},"trusted":true},"execution_count":null,"outputs":[]}]}