{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-15T23:03:07.268143Z","iopub.execute_input":"2022-09-15T23:03:07.268485Z","iopub.status.idle":"2022-09-15T23:03:07.279729Z","shell.execute_reply.started":"2022-09-15T23:03:07.268460Z","shell.execute_reply":"2022-09-15T23:03:07.278228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\nimport gc\n# import cudf\nimport joblib\nimport time\nimport datetime\nfrom tqdm.auto import tqdm\n# from datetime import datetime \nfrom sklearn import metrics\nfrom sklearn.decomposition import PCA\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.metrics import roc_auc_score,roc_curve, auc\nfrom sklearn.metrics import confusion_matrix, precision_score, recall_score\nimport lightgbm as lgb\nfrom sklearn.metrics import RocCurveDisplay","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.281753Z","iopub.execute_input":"2022-09-15T23:03:07.282376Z","iopub.status.idle":"2022-09-15T23:03:07.293560Z","shell.execute_reply.started":"2022-09-15T23:03:07.282311Z","shell.execute_reply":"2022-09-15T23:03:07.292458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DT = joblib.load('../input/amex-lightgbm-models/DT2022-09-15-12_28.model')\nrf = joblib.load('../input/amex-lightgbm-models/RF2022-09-15-12_46.model')\nhgb = joblib.load('../input/amex-lightgbm-models/hgb2022-09-15-13_57.model')\nclf = lgb.Booster(model_file='../input/amex-lightgbm-models/lgb2022-09-14-23_09.model')\nsgd = joblib.load('../input/amex-lightgbm-models/sgd2022-09-15-12_13.model')\nlogi = joblib.load('../input/amex-lightgbm-models/logi2022-09-15-12_05.model')","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.294758Z","iopub.execute_input":"2022-09-15T23:03:07.295002Z","iopub.status.idle":"2022-09-15T23:03:07.785501Z","shell.execute_reply.started":"2022-09-15T23:03:07.294978Z","shell.execute_reply":"2022-09-15T23:03:07.784451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntrain_X = pd.read_feather(\"../input/amex-feather-fe/train_fe.feather\")\nlabel = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\nlabel.sort_values(by = \"customer_ID\",ascending=False)\ntrain_X.sort_values(by = \"customer_ID\",ascending=False)\ny = label.drop(\"customer_ID\",axis = 1)[[\"target\"]]\n\nfeatures = [col for col in train_X.columns if col != \"customer_ID\"]\ncat_features = train_X.select_dtypes(include=['category']).columns.to_list()\nprint(train_X.shape)\n# train_X.head(3)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.789344Z","iopub.execute_input":"2022-09-15T23:03:07.789632Z","iopub.status.idle":"2022-09-15T23:03:07.797354Z","shell.execute_reply.started":"2022-09-15T23:03:07.789609Z","shell.execute_reply":"2022-09-15T23:03:07.796413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_X.isnull().any(axis=0).sum()\n# # turn the rest NA to 0\n# for col in tqdm(features):\n#     train_X[col] = train_X[col].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.798647Z","iopub.execute_input":"2022-09-15T23:03:07.798917Z","iopub.status.idle":"2022-09-15T23:03:07.805794Z","shell.execute_reply.started":"2022-09-15T23:03:07.798894Z","shell.execute_reply":"2022-09-15T23:03:07.804789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Stratify and split \n# X_tra,X_tes,y_tra,y_tes = train_test_split(train_X[features],y,stratify=y,test_size=0.25, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.807241Z","iopub.execute_input":"2022-09-15T23:03:07.807566Z","iopub.status.idle":"2022-09-15T23:03:07.815875Z","shell.execute_reply.started":"2022-09-15T23:03:07.807527Z","shell.execute_reply":"2022-09-15T23:03:07.814951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n# Decomposition\npca = PCA(n_components=100) # retain 95% variance\npca.fit(X_tra) #.iloc[0:200000,:]\npca_train = pca.transform(X_tra)\n\npca_test = pca.transform(X_tes)\n\n# if need to go back to original size\n# approximation = pca.inverse_transform(lower_dimensional_data)\npca.explained_variance_ratio_\n'''","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.816686Z","iopub.execute_input":"2022-09-15T23:03:07.816916Z","iopub.status.idle":"2022-09-15T23:03:07.827865Z","shell.execute_reply.started":"2022-09-15T23:03:07.816892Z","shell.execute_reply":"2022-09-15T23:03:07.827060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nfig, ax = plt.subplots()\n\nmodels = [\n#     (\"LR wiht PCA\", logi),\n    (\"RF\", rf),\n#     (\"SGD LR with PCA\", sgd),\n    (\"DT\", DT),\n#     (\"KNN\", knn),\n#     (\"KNN w/o PCA\", knn2),\n    (\"Hist Gradient Boosting\", hgb),\n#     (\"LightGBM\", clf)\n]\n\nmodel_displays = {}\nfor name, pipeline in models:\n    model_displays[name] = RocCurveDisplay.from_estimator(\n        pipeline, X_tes, y_tes, ax=ax, name=name\n    )\n\nmodels2 = [\n    (\"LR wiht PCA\", logi),\n#     (\"RF\", rf),\n    (\"SGD LR with PCA\", sgd),\n#     (\"DT\", DT),\n#     (\"KNN\", knn),\n#     (\"KNN w/o PCA\", knn2),\n#     (\"Hist Gradient Boosting\", hgb),\n#     (\"LightGBM\", lgb)\n]\n\nmodel_displays2 = {}\nfor name, pipeline in models2:\n    model_displays[name] = RocCurveDisplay.from_estimator(\n        pipeline, pca_test, y_tes, ax=ax, name=name\n    )\n_ = ax.set_title(\"ROC curve\")    \n\nplt.savefig(\"RF CM.png\")\n'''","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.828784Z","iopub.execute_input":"2022-09-15T23:03:07.829081Z","iopub.status.idle":"2022-09-15T23:03:07.839766Z","shell.execute_reply.started":"2022-09-15T23:03:07.829056Z","shell.execute_reply":"2022-09-15T23:03:07.839101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train_X, y, X_tra,X_tes,y_tra,y_tes, label, cat_features\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.842418Z","iopub.execute_input":"2022-09-15T23:03:07.842681Z","iopub.status.idle":"2022-09-15T23:03:07.853820Z","shell.execute_reply.started":"2022-09-15T23:03:07.842658Z","shell.execute_reply":"2022-09-15T23:03:07.853179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import test data\ntest_X = pd.read_feather(\"../input/amex-feather-fe/test_fe.feather\")\ntest_X = test_X.drop(\"index\",axis=1)\n# print(test_X[\"D_64_\"].unique()) # this column can be deleted, with no useful info\n# test_X = test_X.drop(\"D_64_\",axis=1)\nprint(test_X.shape)\n# test_X.head(3)\nIDs = test_X[\"customer_ID\"]\nNUM_ROWS = IDs.shape[0]\ntest_X = test_X.drop(\"customer_ID\",axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:07.855186Z","iopub.execute_input":"2022-09-15T23:03:07.855397Z","iopub.status.idle":"2022-09-15T23:03:13.008689Z","shell.execute_reply.started":"2022-09-15T23:03:07.855375Z","shell.execute_reply":"2022-09-15T23:03:13.007509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# turn the rest NA to 0\n# test_X.isnull().any(axis=0).sum()\n# features = test_X.columns.to_list()\nfor col in tqdm(test_X.columns.to_list()):\n    test_X[col] = test_X[col].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:13.010139Z","iopub.execute_input":"2022-09-15T23:03:13.011279Z","iopub.status.idle":"2022-09-15T23:03:15.654307Z","shell.execute_reply.started":"2022-09-15T23:03:13.011239Z","shell.execute_reply":"2022-09-15T23:03:15.653343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction Prob on test data with HGB\n\nrows = test_X.shape[0]//5\ny_pred_hgb = []\nfor i in tqdm(range(5)):\n    if i != 4:\n        y_pred_test = hgb.predict(test_X.iloc[i*rows:(i+1)*rows,:])\n    else:\n        y_pred_test = hgb.predict(test_X.iloc[i*rows:,:])\n    y_pred_hgb = np.append(y_pred_hgb,y_pred_test)\n\ndel y_pred_test\ngc.collect()\n\nprint(y_pred_hgb.shape)\n# setting threshold to .5\n# for i in range(0, NUM_ROWS):\n#     if y_pred_hgb[i]>=.5:       \n#         y_pred_hgb[i]=1\n#     else:\n#         y_pred_hgb[i]=0","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:15.655568Z","iopub.execute_input":"2022-09-15T23:03:15.655924Z","iopub.status.idle":"2022-09-15T23:03:58.700388Z","shell.execute_reply.started":"2022-09-15T23:03:15.655890Z","shell.execute_reply":"2022-09-15T23:03:58.698838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_hgb = pd.DataFrame({\"customer_ID\": IDs})\nsub_hgb[\"prediction\"] = y_pred_hgb.astype(float)\n# sub[\"prediction\"] = y_test\nsub_hgb.to_csv(f\"submission_hgb_{datetime.datetime.now().strftime('%Y-%m-%d-%H_%M')}.csv\", index=False)\nsub_hgb[:10]","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:03:58.702335Z","iopub.execute_input":"2022-09-15T23:03:58.702883Z","iopub.status.idle":"2022-09-15T23:04:00.696717Z","shell.execute_reply.started":"2022-09-15T23:03:58.702858Z","shell.execute_reply":"2022-09-15T23:04:00.695475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sub_hgb,y_pred_hgb\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:06:24.967447Z","iopub.execute_input":"2022-09-15T23:06:24.967766Z","iopub.status.idle":"2022-09-15T23:06:25.076674Z","shell.execute_reply.started":"2022-09-15T23:06:24.967741Z","shell.execute_reply":"2022-09-15T23:06:25.075725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction Prob on test data with RF\n\nrows = test_X.shape[0]//5\ny_pred_rf = []\nfor i in tqdm(range(5)):\n    if i != 4:\n        y_pred_test = rf.predict(test_X.iloc[i*rows:(i+1)*rows,:])\n    else:\n        y_pred_test = rf.predict(test_X.iloc[i*rows:,:])\n    y_pred_rf = np.append(y_pred_rf,y_pred_test)\n\ndel y_pred_test\ngc.collect()\n\nprint(y_pred_rf.shape)\n# setting threshold to .5\n# for i in range(0, NUM_ROWS ):\n#     if y_pred_rf[i]>=.5:       \n#         y_pred_rf[i]=1\n#     else:\n#         y_pred_rf[i]=0","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:06:33.189968Z","iopub.execute_input":"2022-09-15T23:06:33.190511Z","iopub.status.idle":"2022-09-15T23:07:35.841217Z","shell.execute_reply.started":"2022-09-15T23:06:33.190478Z","shell.execute_reply":"2022-09-15T23:07:35.839853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_rf = pd.DataFrame({\"customer_ID\": IDs})\nsub_rf[\"prediction\"] = y_pred_rf.astype(float)\n# sub[\"prediction\"] = y_test\nsub_rf.to_csv(f\"submission_rf_{datetime.datetime.now().strftime('%Y-%m-%d-%H_%M')}.csv\", index=False)\nsub_rf[:10]","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:08:01.746409Z","iopub.execute_input":"2022-09-15T23:08:01.746727Z","iopub.status.idle":"2022-09-15T23:08:03.747595Z","shell.execute_reply.started":"2022-09-15T23:08:01.746703Z","shell.execute_reply":"2022-09-15T23:08:03.746587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sub_rf,y_pred_rf\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:08:13.628857Z","iopub.execute_input":"2022-09-15T23:08:13.629176Z","iopub.status.idle":"2022-09-15T23:08:13.742331Z","shell.execute_reply.started":"2022-09-15T23:08:13.629151Z","shell.execute_reply":"2022-09-15T23:08:13.741021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction Prob on test data with LightGBM\n\nrows = test_X.shape[0]//5\ny_pred_final = []\n\n# partition\nfor i in tqdm(range(5)):\n    if i != 4:\n        y_pred_test = clf.predict(test_X.iloc[i*rows:(i+1)*rows,:])\n    else:\n        y_pred_test = clf.predict(test_X.iloc[i*rows:,:])\n    y_pred_final = np.append(y_pred_final,y_pred_test)\n\ndel test_X,y_pred_test\ngc.collect()\n\nprint(y_pred_final.shape)\n# setting threshold to .5\n# for i in range(0, NUM_ROWS):\n#     if y_pred_final[i]>=.5:       \n#         y_pred_final[i]=1\n#     else:\n#         y_pred_final[i]=0","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:08:17.966568Z","iopub.execute_input":"2022-09-15T23:08:17.966901Z","iopub.status.idle":"2022-09-15T23:08:37.096868Z","shell.execute_reply.started":"2022-09-15T23:08:17.966876Z","shell.execute_reply":"2022-09-15T23:08:37.095767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_lgb = pd.DataFrame({\"customer_ID\": IDs})\nsub_lgb[\"prediction\"] = y_pred_final.astype(float)\n# sub[\"prediction\"] = y_test\nsub_lgb.to_csv(f\"submission_lgbm_{datetime.datetime.now().strftime('%Y-%m-%d-%H_%M')}.csv\", index=False)\nsub_lgb[:10]","metadata":{"execution":{"iopub.status.busy":"2022-09-15T23:08:52.571865Z","iopub.execute_input":"2022-09-15T23:08:52.572206Z","iopub.status.idle":"2022-09-15T23:08:55.188561Z","shell.execute_reply.started":"2022-09-15T23:08:52.572181Z","shell.execute_reply":"2022-09-15T23:08:55.187542Z"},"trusted":true},"execution_count":null,"outputs":[]}]}