{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-27T13:04:49.598062Z","iopub.execute_input":"2024-05-27T13:04:49.598803Z","iopub.status.idle":"2024-05-27T13:04:49.606644Z","shell.execute_reply.started":"2024-05-27T13:04:49.598766Z","shell.execute_reply":"2024-05-27T13:04:49.605426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport time","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:04:49.608905Z","iopub.execute_input":"2024-05-27T13:04:49.609991Z","iopub.status.idle":"2024-05-27T13:04:49.622937Z","shell.execute_reply.started":"2024-05-27T13:04:49.609947Z","shell.execute_reply":"2024-05-27T13:04:49.621390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start=time.time() # how long does it take?\n\nskiprows = list(range(1,1))\ntrain_call_rows = 400000\ntest_call_rows = 407500    # goal : 32552 rows after groupby\ntarget_call_rows = 33164\nprint(\"size of skiprows:\",len(skiprows))\ndf_path = \"/kaggle/input/amex-default-prediction/train_data.csv\"\ntarget_path = \"/kaggle/input/amex-default-prediction/train_labels.csv\"\ntest_path = \"/kaggle/input/amex-default-prediction/test_data.csv\"\n\ndf = pd.read_csv(df_path,skiprows=skiprows,nrows=train_call_rows)\ntarget = pd.read_csv(target_path,skiprows=skiprows,nrows=target_call_rows)\ntest = pd.read_csv(test_path,skiprows=skiprows,nrows=test_call_rows)\n\nend = time.time() - start\nprint(f'time : {end}')\nprint(\"done\")","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:22:32.526759Z","iopub.execute_input":"2024-05-27T13:22:32.527188Z","iopub.status.idle":"2024-05-27T13:23:06.010156Z","shell.execute_reply.started":"2024-05-27T13:22:32.527157Z","shell.execute_reply":"2024-05-27T13:23:06.008511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Groupby customer ID","metadata":{}},{"cell_type":"code","source":"df_grouped = df.groupby(\"customer_ID\").tail(1)\ndf_grouped.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:47:03.331489Z","iopub.execute_input":"2024-05-27T13:47:03.331921Z","iopub.status.idle":"2024-05-27T13:47:03.545021Z","shell.execute_reply.started":"2024-05-27T13:47:03.331888Z","shell.execute_reply":"2024-05-27T13:47:03.543559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Delete columns","metadata":{}},{"cell_type":"code","source":"def get_del_col(df, call_rows):\n    del_col = []\n    for col in df.columns:\n        count = df[col].isna().sum()\n        if count > call_rows*0.5:\n            del_col.append(col)\n    return del_col\n\ntrain_del_col = get_del_col(df, train_call_rows)\ntest_del_col = get_del_col(test, test_call_rows)\n\ndel_col = [col for col in train_del_col or test_del_col]\ndel_col.extend([\"S_2\"]) # now we have del_col + date\nprint(f'del_col_for_train : {len(train_del_col)}')\nprint(f'del_col_for_test : {len(test_del_col)}')\nprint(f'size of del_col : {len(del_col)}')","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:12:16.637026Z","iopub.execute_input":"2024-05-27T13:12:16.637964Z","iopub.status.idle":"2024-05-27T13:12:17.272263Z","shell.execute_reply.started":"2024-05-27T13:12:16.637923Z","shell.execute_reply":"2024-05-27T13:12:17.270949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_reduced = df_grouped.drop(columns = del_col)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:12:18.362101Z","iopub.execute_input":"2024-05-27T13:12:18.363422Z","iopub.status.idle":"2024-05-27T13:12:18.384019Z","shell.execute_reply.started":"2024-05-27T13:12:18.363372Z","shell.execute_reply":"2024-05-27T13:12:18.382441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"obj_col = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nobj_col = [i for i in obj_col if i not in del_col]\n\ndf_hot = pd.get_dummies(data=df_reduced, columns = obj_col)\ndf_hot.info() # bool(43), float64(147), int64(1), object(1)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:12:20.466755Z","iopub.execute_input":"2024-05-27T13:12:20.467326Z","iopub.status.idle":"2024-05-27T13:12:20.543690Z","shell.execute_reply.started":"2024-05-27T13:12:20.467276Z","shell.execute_reply":"2024-05-27T13:12:20.542473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill N/A","metadata":{}},{"cell_type":"code","source":"df_hot.fillna(method=\"ffill\", inplace=True)\ndf_hot.fillna(method=\"bfill\", inplace=True)\ndf_hot.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:39:00.501737Z","iopub.execute_input":"2024-05-27T13:39:00.502482Z","iopub.status.idle":"2024-05-27T13:39:00.528028Z","shell.execute_reply.started":"2024-05-27T13:39:00.502437Z","shell.execute_reply":"2024-05-27T13:39:00.526323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Same to test (groupby -> col del -> one-hot -> fill N/A )","metadata":{}},{"cell_type":"code","source":"test_grouped = test.groupby(\"customer_ID\").tail(1)\ntest_reduced = test_grouped.drop(columns = del_col)\ntest_hot = pd.get_dummies(data=test_reduced, columns = obj_col)\n\ntest_hot.fillna(method=\"ffill\", inplace=True)\ntest_hot.fillna(method=\"bfill\", inplace=True)\ntest_hot.info() # 3???? rows \ndf_hot.info()  # 33164 rows\n# df has 43 bool cols, whereas test has 41 bool cols wtf\n# In test, there aren't [ D_64_-1,D_68_0.0 ]\n# Solve by increasing test rows?\n#  -> solved by rearranging order of process\n# new problem : different rows by test and df (not same people in the same amount of rows fuck)\n#  -> manually increasing test df? -> solved, now we got 33164 x 189 matrix df_hot, test_hot","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:52:22.037472Z","iopub.execute_input":"2024-05-27T13:52:22.038003Z","iopub.status.idle":"2024-05-27T13:52:22.390268Z","shell.execute_reply.started":"2024-05-27T13:52:22.037966Z","shell.execute_reply":"2024-05-27T13:52:22.388844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df_hot\ny_train = target\nX_test = test_hot\n\nprint(X_train.shape)\nprint(Y_train.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:52:25.311188Z","iopub.execute_input":"2024-05-27T13:52:25.312083Z","iopub.status.idle":"2024-05-27T13:52:25.319754Z","shell.execute_reply.started":"2024-05-27T13:52:25.312043Z","shell.execute_reply":"2024-05-27T13:52:25.318235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\n\nxgb = XGBClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\ndt = DecisionTreeClassifier(max_depth = 1, random_state = 31)\nlgbm = LGBMClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\nada = AdaBoostClassifier(base_estimator = dt, n_estimators = 100, learning_rate = 1, random_state = 31) ##ada는 무조건 dt랑 같이 쓰나?\nrf = RandomForestClassifier(max_depth = 1, n_estimators = 100, random_state = 31)\ngb = GradientBoostingClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\nbag = BaggingClassifier(n_estimators = 100, random_state = 31)\nknn = KNeighborsClassifier(n_neighbors = 3)\nvote_hard = VotingClassifier(estimators = [('xgb', xgb), ('dt', dt), ('lgbm', lgbm), ('ada', ada), ('rf', rf), ('gb', gb), ('bag', bag), ('knn', knn)], voting = 'hard')\nvote_soft = VotingClassifier(estimators = [('xgb', xgb), ('dt', dt), ('lgbm', lgbm), ('ada', ada), ('rf', rf), ('gb', gb), ('bag', bag), ('knn', knn)], voting = 'soft')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:52:28.296592Z","iopub.execute_input":"2024-05-27T13:52:28.297304Z","iopub.status.idle":"2024-05-27T13:52:28.304505Z","shell.execute_reply.started":"2024-05-27T13:52:28.297269Z","shell.execute_reply":"2024-05-27T13:52:28.303077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb.fit(X_train, y_train) # float not working damn\ndt.fit(X_train, y_train)\nlgbm.fit(X_train, y_train)\nada.fit(X_train, y_train)\nrf.fit(X_train, y_train)\ngb.fit(X_train, y_train)\nbag.fit(X_train, y_train)\nknn.fit(X_train, y_train)\nvote_hard.fit(X_train, y_train)\nvote_soft.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:56:22.049500Z","iopub.execute_input":"2024-05-27T13:56:22.049953Z","iopub.status.idle":"2024-05-27T13:56:22.198553Z","shell.execute_reply.started":"2024-05-27T13:56:22.049912Z","shell.execute_reply":"2024-05-27T13:56:22.196719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_xgb = xgb.predict(X_test)\ny_pred_dt = dt.predict(X_test)\ny_pred_lgbm = lgbm.predict(X_test)\ny_pred_ada = ada.predict(X_test)\ny_pred_rf = rf.predict(X_test)\ny_pred_gb = gb.predict(X_test)\ny_pred_bag = bag.predict(X_test)\ny_pred_knn = knn.predict(X_test)\ny_pred_vote_hard = vote_hard.predict(X_test)\ny_pred_vote_soft = vote_soft.predict(X_test)\ny_preds = [y_pred_xgb,y_pred_dt,y_pred_lgbm,y_pred_ada,y_pred_rf,y_pred_gb,y_pred_bag,y_pred_vote_hard,y_pred_vote_soft]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check Accuracy","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for pred in y_preds:\n    accuracy = amex_metric(y_test, pred)\n    print(f\"xgb accuracy: {accuracy:.7f}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Matrix ( Visualizing )","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\ncm = confusion_matrix(y_test, y_pred_bag)\nplt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]}]}