{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3739819,"sourceType":"datasetVersion","datasetId":2231132}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-28T10:17:38.578056Z","iopub.execute_input":"2024-05-28T10:17:38.578535Z","iopub.status.idle":"2024-05-28T10:17:38.600924Z","shell.execute_reply.started":"2024-05-28T10:17:38.578490Z","shell.execute_reply":"2024-05-28T10:17:38.599986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport time","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:38.603023Z","iopub.execute_input":"2024-05-28T10:17:38.603689Z","iopub.status.idle":"2024-05-28T10:17:38.608308Z","shell.execute_reply.started":"2024-05-28T10:17:38.603651Z","shell.execute_reply":"2024-05-28T10:17:38.607218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start=time.time() # how long does it take?\n\ndf_path =\"/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet\"\ntarget_path = \"/kaggle/input/amex-default-prediction/train_labels.csv\"\ntest_path = \"/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet\"\nsample_path = \"/kaggle/input/amex-default-prediction/sample_submission.csv\"\n\ndf = pd.read_parquet(path=df_path)\nend1 = time.time() - start\n\ntest = pd.read_parquet(path=test_path)\nend2 = time.time() - start - end1\n\ntarget = pd.read_csv(target_path)\nend3 = time.time() - start - end1 - end2\n\nsample = pd.read_csv(sample_path)\nend4 = time.time() - start - end1 - end2 - end3\nprint(f'time : {end1}')\nprint(f'time : {end2}')\nprint(f'time : {end3}')\nprint(f'time : {end4}')\nprint(\"done\")","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:17:38.609977Z","iopub.execute_input":"2024-05-28T10:17:38.611026Z","iopub.status.idle":"2024-05-28T10:18:08.607280Z","shell.execute_reply.started":"2024-05-28T10:17:38.610986Z","shell.execute_reply":"2024-05-28T10:18:08.606154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Groupby customer ID","metadata":{}},{"cell_type":"code","source":"df[\"B_38\"].value_counts().index[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:18:08.608834Z","iopub.execute_input":"2024-05-28T10:18:08.609447Z","iopub.status.idle":"2024-05-28T10:18:08.653372Z","shell.execute_reply.started":"2024-05-28T10:18:08.609414Z","shell.execute_reply":"2024-05-28T10:18:08.651817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:50:07.907263Z","iopub.status.idle":"2024-05-28T10:50:07.907789Z","shell.execute_reply.started":"2024-05-28T10:50:07.907527Z","shell.execute_reply":"2024-05-28T10:50:07.907547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\nagg_dict = {}\nobj_col = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnot_obj = [i for i in df.columns if (i not in obj_col)]\nnot_obj.remove(\"customer_ID\")\nnot_obj.remove(\"S_2\")\nfor col in obj_col:\n    agg_dict[col] = lambda x: x.value_counts().index[0]\nfor col in not_obj:\n    agg_dict[col] = 'mean'\ndf_grouped = df.groupby(\"customer_ID\").agg(agg_dict)\n\nend = time.time()-start\nprint(end)\nprint(df_grouped)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:50:25.811257Z","iopub.execute_input":"2024-05-28T10:50:25.811673Z","iopub.status.idle":"2024-05-28T10:52:33.611051Z","shell.execute_reply.started":"2024-05-28T10:50:25.811641Z","shell.execute_reply":"2024-05-28T10:52:33.608701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_grouped = df.groupby(\"customer_ID\").tail(1)\ndf_grouped.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.474468Z","iopub.status.idle":"2024-05-28T10:19:38.475650Z","shell.execute_reply.started":"2024-05-28T10:19:38.475299Z","shell.execute_reply":"2024-05-28T10:19:38.475335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Delete columns","metadata":{}},{"cell_type":"code","source":"def get_del_col(df, call_rows):\n    del_col = []\n    for col in df.columns:\n        count = df[col].isna().sum()\n        if count > call_rows*0.5:\n            del_col.append(col)\n    return del_col\n\ntrain_del_col = get_del_col(df, train_call_rows)\ntest_del_col = get_del_col(test, test_call_rows)\ndel_col = list(set(train_del_col) | set(test_del_col))\ndel_col.extend([\"S_2\"]) # now we have del_col + date\nprint(f'del_col_for_train : {len(train_del_col)}')\nprint(f'del_col_for_test : {len(test_del_col)}')\nprint(f'size of del_col : {len(del_col)}')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.478156Z","iopub.status.idle":"2024-05-28T10:19:38.478576Z","shell.execute_reply.started":"2024-05-28T10:19:38.478382Z","shell.execute_reply":"2024-05-28T10:19:38.478398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_reduced = df_grouped.drop(columns = del_col)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.480432Z","iopub.status.idle":"2024-05-28T10:19:38.480867Z","shell.execute_reply.started":"2024-05-28T10:19:38.480644Z","shell.execute_reply":"2024-05-28T10:19:38.480660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"obj_col = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nobj_col = [i for i in obj_col if i not in del_col]\n\n\ndf_hot = pd.get_dummies(data=df_reduced, columns = obj_col) \n\n# bool(50), float32(59), int16(9), int8(75), object(1)\n\n# change boolean col -> int col\n#bool_col = df_hot.select_dtypes(\"bool\").columns\n#for col in bool_col:\n#    df_hot[col] = df_hot[col].astype(\"int8\")\n    \ndf_hot.info()   # float32(59), int16(9), int8(125), object(1)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.482121Z","iopub.status.idle":"2024-05-28T10:19:38.482501Z","shell.execute_reply.started":"2024-05-28T10:19:38.482314Z","shell.execute_reply":"2024-05-28T10:19:38.482329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill N/A","metadata":{}},{"cell_type":"code","source":"df_hot.fillna(method=\"ffill\", inplace=True)\ndf_hot.fillna(method=\"bfill\", inplace=True)\ndf_hot.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.483609Z","iopub.status.idle":"2024-05-28T10:19:38.484008Z","shell.execute_reply.started":"2024-05-28T10:19:38.483815Z","shell.execute_reply":"2024-05-28T10:19:38.483830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Same to test (groupby -> col del -> one-hot -> fill N/A )","metadata":{}},{"cell_type":"code","source":"test_grouped = test.groupby(\"customer_ID\").tail(1)\ntest_reduced = test_grouped.drop(columns = del_col)\ntest_hot = pd.get_dummies(data=test_reduced, columns = obj_col)\n\ntest_hot.fillna(method=\"ffill\", inplace=True)\ntest_hot.fillna(method=\"bfill\", inplace=True)\ntest_hot.info() # 3???? rows \ndf_hot.info()  # 33164 rows\n# df has 43 bool cols, whereas test has 41 bool cols wtf\n# In test, there aren't [ D_64_-1,D_68_0.0 ]\n# Solve by increasing test rows?\n#  -> solved by rearranging order of process\n# new problem : different rows by test and df (not same people in the same amount of rows fuck)\n#  -> manually increasing test df? -> solved, now we got 33164 x 189 matrix df_hot, test_hot","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.485439Z","iopub.status.idle":"2024-05-28T10:19:38.486224Z","shell.execute_reply.started":"2024-05-28T10:19:38.485997Z","shell.execute_reply":"2024-05-28T10:19:38.486017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hotter = df_hot.set_index(\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.487945Z","iopub.status.idle":"2024-05-28T10:19:38.488339Z","shell.execute_reply.started":"2024-05-28T10:19:38.488147Z","shell.execute_reply":"2024-05-28T10:19:38.488163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bool_col = df_hotter.select_dtypes(\"bool\").columns\nfor col in bool_col:\n    df_hotter[col] = df_hotter[col].astype(\"int8\")\n    \ndf_hotter.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.489838Z","iopub.status.idle":"2024-05-28T10:19:38.490231Z","shell.execute_reply.started":"2024-05-28T10:19:38.490030Z","shell.execute_reply":"2024-05-28T10:19:38.490056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hotter.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.491823Z","iopub.status.idle":"2024-05-28T10:19:38.492418Z","shell.execute_reply.started":"2024-05-28T10:19:38.492206Z","shell.execute_reply":"2024-05-28T10:19:38.492225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df_hotter\ny_train = target[\"target\"]\nX_test = test_hot\n\nprint(X_train.shape)\nprint(y_train.shape)\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.493621Z","iopub.status.idle":"2024-05-28T10:19:38.494023Z","shell.execute_reply.started":"2024-05-28T10:19:38.493828Z","shell.execute_reply":"2024-05-28T10:19:38.493844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\n\nxgb = XGBClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\ndt = DecisionTreeClassifier(max_depth = 1, random_state = 31)\nlgbm = LGBMClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\nada = AdaBoostClassifier(base_estimator = dt, n_estimators = 100, learning_rate = 1, random_state = 31) ##ada는 무조건 dt랑 같이 쓰나?\nrf = RandomForestClassifier(max_depth = 1, n_estimators = 100, random_state = 31)\ngb = GradientBoostingClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\nbag = BaggingClassifier(n_estimators = 100, random_state = 31)\nknn = KNeighborsClassifier(n_neighbors = 3)\nvote_hard = VotingClassifier(estimators = [('xgb', xgb), ('dt', dt), ('lgbm', lgbm), ('gb', gb), ('bag', bag)], voting = 'hard')\nvote_soft = VotingClassifier(estimators = [('xgb', xgb), ('dt', dt), ('lgbm', lgbm), ('gb', gb), ('bag', bag)], voting = 'soft')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.495273Z","iopub.status.idle":"2024-05-28T10:19:38.495645Z","shell.execute_reply.started":"2024-05-28T10:19:38.495458Z","shell.execute_reply":"2024-05-28T10:19:38.495473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\nxgb.fit(X_train, y_train)\ndt.fit(X_train, y_train)\nlgbm.fit(X_train, y_train)\ngb.fit(X_train, y_train)\nbag.fit(X_train, y_train)\nend = time.time()- start\nprint(end)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.498241Z","iopub.status.idle":"2024-05-28T10:19:38.498619Z","shell.execute_reply.started":"2024-05-28T10:19:38.498433Z","shell.execute_reply":"2024-05-28T10:19:38.498449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\nvote_hard.fit(X_train, y_train)\nvote_soft.fit(X_train, y_train)\n\nend = time.time()- start\nprint(end)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.500028Z","iopub.status.idle":"2024-05-28T10:19:38.500442Z","shell.execute_reply.started":"2024-05-28T10:19:38.500248Z","shell.execute_reply":"2024-05-28T10:19:38.500264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start=time.time()\n#ada.fit(X_train, y_train)\n#rf.fit(X_train, y_train)\n#knn.fit(X_train, y_train)\n\n\n\nend = time.time()- start\nprint(end)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.501921Z","iopub.status.idle":"2024-05-28T10:19:38.502305Z","shell.execute_reply.started":"2024-05-28T10:19:38.502111Z","shell.execute_reply":"2024-05-28T10:19:38.502133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_xgb = xgb.predict(X_test)\ny_pred_dt = dt.predict(X_test)\ny_pred_lgbm = lgbm.predict(X_test)\ny_pred_gb = gb.predict(X_test)\ny_pred_bag = bag.predict(X_test)\ny_pred_vote_hard = vote_hard.predict(X_test)\ny_pred_vote_soft = vote_soft.predict(X_test)\ny_preds = [y_pred_xgb,y_pred_dt,y_pred_lgbm,y_pred_gb,y_pred_bag,y_pred_vote_hard,y_pred_vote_soft]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.504088Z","iopub.status.idle":"2024-05-28T10:19:38.504482Z","shell.execute_reply.started":"2024-05-28T10:19:38.504288Z","shell.execute_reply":"2024-05-28T10:19:38.504304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check Accuracy","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.505958Z","iopub.status.idle":"2024-05-28T10:19:38.506426Z","shell.execute_reply.started":"2024-05-28T10:19:38.506224Z","shell.execute_reply":"2024-05-28T10:19:38.506240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for pred in y_preds:\n    accuracy = amex_metric(y_test, pred)\n    print(f\"xgb accuracy: {accuracy:.7f}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.507605Z","iopub.status.idle":"2024-05-28T10:19:38.507982Z","shell.execute_reply.started":"2024-05-28T10:19:38.507798Z","shell.execute_reply":"2024-05-28T10:19:38.507813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submisison","metadata":{}},{"cell_type":"code","source":"submit = y_pred_...            # choose this\nsubmit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.509205Z","iopub.status.idle":"2024-05-28T10:19:38.509570Z","shell.execute_reply.started":"2024-05-28T10:19:38.509389Z","shell.execute_reply":"2024-05-28T10:19:38.509404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Matrix ( Visualizing )","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\ncm = confusion_matrix(y_test, y_pred_bag)\nplt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:19:38.510668Z","iopub.status.idle":"2024-05-28T10:19:38.511091Z","shell.execute_reply.started":"2024-05-28T10:19:38.510888Z","shell.execute_reply":"2024-05-28T10:19:38.510904Z"},"trusted":true},"execution_count":null,"outputs":[]}]}