{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3739819,"sourceType":"datasetVersion","datasetId":2231132}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-25T09:01:14.212175Z","iopub.execute_input":"2024-06-25T09:01:14.212584Z","iopub.status.idle":"2024-06-25T09:01:14.226217Z","shell.execute_reply.started":"2024-06-25T09:01:14.212548Z","shell.execute_reply":"2024-06-25T09:01:14.224721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport time","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:01:16.520668Z","iopub.execute_input":"2024-06-25T09:01:16.521134Z","iopub.status.idle":"2024-06-25T09:01:16.526415Z","shell.execute_reply.started":"2024-06-25T09:01:16.521101Z","shell.execute_reply":"2024-06-25T09:01:16.525324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start=time.time() # how long does it take?\n\ndf_path =\"/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet\"\ntarget_path = \"/kaggle/input/amex-default-prediction/train_labels.csv\"\ntest_path = \"/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet\"\nsample_path = \"/kaggle/input/amex-default-prediction/sample_submission.csv\"\n\ndf = pd.read_parquet(path=df_path)\nend1 = time.time() - start\n\ntest = pd.read_parquet(path=test_path)\nend2 = time.time() - start - end1\n\ntarget = pd.read_csv(target_path)\nend3 = time.time() - start - end1 - end2\n\nsample = pd.read_csv(sample_path)\nend4 = time.time() - start - end1 - end2 - end3\nprint(f'time : {end1}')\nprint(f'time : {end2}')\nprint(f'time : {end3}')\nprint(f'time : {end4}')\nprint(\"done\")","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:01:21.355121Z","iopub.execute_input":"2024-06-25T09:01:21.355575Z","iopub.status.idle":"2024-06-25T09:02:20.952938Z","shell.execute_reply.started":"2024-06-25T09:01:21.355538Z","shell.execute_reply":"2024-06-25T09:02:20.951779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Order (groupby -> col del -> one-hot -> fill N/A )","metadata":{}},{"cell_type":"markdown","source":"## Groupby customer ID","metadata":{}},{"cell_type":"code","source":"start = time.time()\nagg_dict = {}\nobj_col = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnot_obj = [i for i in df.columns if (i not in obj_col)]\nnot_obj.remove(\"customer_ID\")\nnot_obj.remove(\"S_2\")\n\ndf_grouped = df.groupby(\"customer_ID\").tail(1)\ntest_grouped = test.groupby(\"customer_ID\").tail(1)\nend = time.time()-start\nprint(end)\n","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-06-25T09:02:29.670523Z","iopub.execute_input":"2024-06-25T09:02:29.671597Z","iopub.status.idle":"2024-06-25T09:02:37.621676Z","shell.execute_reply.started":"2024-06-25T09:02:29.67149Z","shell.execute_reply":"2024-06-25T09:02:37.620505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Delete columns","metadata":{}},{"cell_type":"code","source":"df_grouped.info()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:02:49.288569Z","iopub.execute_input":"2024-06-25T09:02:49.288957Z","iopub.status.idle":"2024-06-25T09:02:49.332753Z","shell.execute_reply.started":"2024-06-25T09:02:49.288929Z","shell.execute_reply":"2024-06-25T09:02:49.331306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_grouped.info()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:02:58.644712Z","iopub.execute_input":"2024-06-25T09:02:58.645116Z","iopub.status.idle":"2024-06-25T09:02:58.663673Z","shell.execute_reply.started":"2024-06-25T09:02:58.645084Z","shell.execute_reply":"2024-06-25T09:02:58.662451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_del_col(df, call_rows):\n    del_col = []\n    for col in df.columns:\n        count = df[col].isna().sum()\n        if count > call_rows*0.5:\n            del_col.append(col)\n    return del_col\n\ntrain_del_col = get_del_col(df, 458913)\ntest_del_col = get_del_col(test, 924621)\ndel_col = list(set(train_del_col) | set(test_del_col))\ndel_col.extend([\"S_2\"]) # now we have del_col + date\nprint(f'del_col_for_train : {len(train_del_col)}')\nprint(f'del_col_for_test : {len(test_del_col)}')\nprint(f'size of del_col : {len(del_col)}')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:03:26.695013Z","iopub.execute_input":"2024-06-25T09:03:26.695424Z","iopub.status.idle":"2024-06-25T09:03:31.419595Z","shell.execute_reply.started":"2024-06-25T09:03:26.695368Z","shell.execute_reply":"2024-06-25T09:03:31.418518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_reduced = df_grouped.drop(columns = del_col)\ntest_reduced = test_grouped.drop(columns = del_col)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:06:05.285644Z","iopub.execute_input":"2024-06-25T09:06:05.286079Z","iopub.status.idle":"2024-06-25T09:06:05.733987Z","shell.execute_reply.started":"2024-06-25T09:06:05.286046Z","shell.execute_reply":"2024-06-25T09:06:05.732076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One-hot encoding","metadata":{}},{"cell_type":"code","source":"obj_col = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nobj_col = [i for i in obj_col if i not in del_col]\n\n\ndf_hot = pd.get_dummies(data=df_reduced, columns = obj_col) \ntest_hot = pd.get_dummies(data=test_reduced, columns = obj_col) \n\n# bool(50), float32(59), int16(9), int8(75), object(1)\n\ndf_hot.info()   # float32(59), int16(9), int8(125), object(1)\ntest_hot.info()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:06:26.8007Z","iopub.execute_input":"2024-06-25T09:06:26.80108Z","iopub.status.idle":"2024-06-25T09:06:27.877905Z","shell.execute_reply.started":"2024-06-25T09:06:26.801049Z","shell.execute_reply":"2024-06-25T09:06:27.876783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill N/A","metadata":{}},{"cell_type":"code","source":"df_hot.fillna(method=\"ffill\", inplace=True)\ndf_hot.fillna(method=\"bfill\", inplace=True)\ndf_hot.isna().sum().sum()\n\ntest_hot.fillna(method=\"ffill\", inplace=True)\ntest_hot.fillna(method=\"bfill\", inplace=True)\ntest_hot.isna().sum().sum()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:07:11.11836Z","iopub.execute_input":"2024-06-25T09:07:11.118785Z","iopub.status.idle":"2024-06-25T09:07:12.271964Z","shell.execute_reply.started":"2024-06-25T09:07:11.118752Z","shell.execute_reply":"2024-06-25T09:07:12.270497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Boolean column -> int8 type column","metadata":{}},{"cell_type":"code","source":"df_hotter = df_hot.set_index(\"customer_ID\")\ntest_hotter = test_hot.set_index(\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:09:53.518651Z","iopub.execute_input":"2024-06-25T09:09:53.519054Z","iopub.status.idle":"2024-06-25T09:09:53.975226Z","shell.execute_reply.started":"2024-06-25T09:09:53.519023Z","shell.execute_reply":"2024-06-25T09:09:53.973969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bool_col = df_hotter.select_dtypes(\"bool\").columns\nfor col in bool_col:\n    df_hotter[col] = df_hotter[col].astype(\"int8\")\n    test_hotter[col] = test_hotter[col].astype(\"int8\")\n\n    \ndf_hotter.info()\ntest_hotter.info()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:10:51.364126Z","iopub.execute_input":"2024-06-25T09:10:51.364549Z","iopub.status.idle":"2024-06-25T09:10:51.394852Z","shell.execute_reply.started":"2024-06-25T09:10:51.364515Z","shell.execute_reply":"2024-06-25T09:10:51.393365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hotter.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:11:06.61998Z","iopub.execute_input":"2024-06-25T09:11:06.620367Z","iopub.status.idle":"2024-06-25T09:11:06.64857Z","shell.execute_reply.started":"2024-06-25T09:11:06.620338Z","shell.execute_reply":"2024-06-25T09:11:06.647545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(df_hotter,target[\"target\"],test_size=0.2,random_state=31)\n\n\nprint(X_train.shape)\nprint(y_train.shape)\nprint(X_test.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:54:04.675713Z","iopub.execute_input":"2024-06-25T09:54:04.676122Z","iopub.status.idle":"2024-06-25T09:54:05.442981Z","shell.execute_reply.started":"2024-06-25T09:54:04.676092Z","shell.execute_reply":"2024-06-25T09:54:05.44153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"markdown","source":"## Fitting","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom catboost import CatBoostClassifier, Pool\n\ncat = CatBoostClassifier(iterations=2, learning_rate=1, depth=2,loss_function='Logloss', verbose=True)\nxgb = XGBClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\ndt = DecisionTreeClassifier(max_depth = 1, random_state = 31)\nlgbm = LGBMClassifier(force_col_wise=True, max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\nada = AdaBoostClassifier(base_estimator = dt, n_estimators = 100, learning_rate = 1, random_state = 31) ##ada는 무조건 dt랑 같이 쓰나?\nrf = RandomForestClassifier(max_depth = 1, n_estimators = 100, random_state = 31)\ngb = GradientBoostingClassifier(max_depth = 1, n_estimators = 100, learning_rate = 1, random_state = 31)\nbag = BaggingClassifier(n_estimators = 100, random_state = 31)\nknn = KNeighborsClassifier(n_neighbors = 3)\nvote_hard = VotingClassifier(estimators = [('xgb', xgb), ('dt', dt), ('lgbm', lgbm), ('cat',cat)], voting = 'hard')\nvote_soft = VotingClassifier(estimators = [('xgb', xgb), ('dt', dt), ('lgbm', lgbm), ('cat',cat)], voting = 'soft')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:54:09.224962Z","iopub.execute_input":"2024-06-25T09:54:09.225359Z","iopub.status.idle":"2024-06-25T09:54:09.236937Z","shell.execute_reply.started":"2024-06-25T09:54:09.225329Z","shell.execute_reply":"2024-06-25T09:54:09.235736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\n\nxgb.fit(X_train, y_train) # fast\ndt.fit(X_train, y_train) # fast\nlgbm.fit(X_train, y_train) # mid \ncat.fit(X_train, y_train) # fast\n# 93 sec\nend = time.time()- start\nprint(end) # output time","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:54:15.29699Z","iopub.execute_input":"2024-06-25T09:54:15.297453Z","iopub.status.idle":"2024-06-25T09:55:29.555937Z","shell.execute_reply.started":"2024-06-25T09:54:15.297409Z","shell.execute_reply":"2024-06-25T09:55:29.554669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\nvote_hard.fit(X_train, y_train)\nvote_soft.fit(X_train, y_train) # 오래걸림 3 min\n\nend = time.time()- start\nprint(end)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:55:33.998307Z","iopub.execute_input":"2024-06-25T09:55:33.998756Z","iopub.status.idle":"2024-06-25T09:58:01.884708Z","shell.execute_reply.started":"2024-06-25T09:55:33.998722Z","shell.execute_reply":"2024-06-25T09:58:01.883325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict","metadata":{}},{"cell_type":"code","source":"y_pred_xgb = xgb.predict(X_test)\ny_pred_dt = dt.predict(X_test)\ny_pred_lgbm = lgbm.predict(X_test)\ny_pred_cat = cat.predict(X_test)\ny_pred_vote_hard = vote_hard.predict(X_test)\ny_pred_vote_soft = vote_soft.predict(X_test)\n\ny_preds = [y_pred_xgb,\n           y_pred_dt,\n           y_pred_lgbm,\n           y_pred_cat,\n           y_pred_vote_hard,\n           y_pred_vote_soft]","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:58:06.634555Z","iopub.execute_input":"2024-06-25T09:58:06.635004Z","iopub.status.idle":"2024-06-25T09:58:09.306779Z","shell.execute_reply.started":"2024-06-25T09:58:06.634969Z","shell.execute_reply":"2024-06-25T09:58:09.305569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check Accuracy","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T09:58:11.610382Z","iopub.execute_input":"2024-06-25T09:58:11.610826Z","iopub.status.idle":"2024-06-25T09:58:11.624498Z","shell.execute_reply.started":"2024-06-25T09:58:11.610793Z","shell.execute_reply":"2024-06-25T09:58:11.623269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nfor pred in y_preds:\n    accuracy = accuracy_score(y_test, pred)\n    print(f\"accuracy: {accuracy:.7f}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-25T10:01:48.908766Z","iopub.execute_input":"2024-06-25T10:01:48.909198Z","iopub.status.idle":"2024-06-25T10:01:48.977568Z","shell.execute_reply.started":"2024-06-25T10:01:48.909161Z","shell.execute_reply":"2024-06-25T10:01:48.976305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submisison","metadata":{}},{"cell_type":"code","source":"y_pred_goal = xgb.predict(test_hotter) # choose this","metadata":{"execution":{"iopub.status.busy":"2024-06-25T10:23:01.884359Z","iopub.execute_input":"2024-06-25T10:23:01.885517Z","iopub.status.idle":"2024-06-25T10:23:05.094556Z","shell.execute_reply.started":"2024-06-25T10:23:01.885475Z","shell.execute_reply":"2024-06-25T10:23:05.093497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample[\"prediction\"] = y_pred_goal  \nsub_df = sample[['customer_ID', 'prediction']].copy()\nsub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-25T10:22:43.383383Z","iopub.execute_input":"2024-06-25T10:22:43.383829Z","iopub.status.idle":"2024-06-25T10:22:45.747561Z","shell.execute_reply.started":"2024-06-25T10:22:43.383797Z","shell.execute_reply":"2024-06-25T10:22:45.746362Z"},"trusted":true},"execution_count":null,"outputs":[]}]}