{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-03T06:33:41.143147Z","iopub.execute_input":"2022-06-03T06:33:41.143659Z","iopub.status.idle":"2022-06-03T06:33:41.184015Z","shell.execute_reply.started":"2022-06-03T06:33:41.143564Z","shell.execute_reply":"2022-06-03T06:33:41.183071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install autogluon","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-06-03T06:33:41.185647Z","iopub.execute_input":"2022-06-03T06:33:41.186584Z","iopub.status.idle":"2022-06-03T06:36:36.138497Z","shell.execute_reply.started":"2022-06-03T06:33:41.186538Z","shell.execute_reply":"2022-06-03T06:36:36.137243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_pickle('../input/amex-data-pckl-files/train_agg.pkl', compression='gzip')\ntrain_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:36.141713Z","iopub.execute_input":"2022-06-03T06:36:36.142712Z","iopub.status.idle":"2022-06-03T06:36:47.784366Z","shell.execute_reply.started":"2022-06-03T06:36:36.142646Z","shell.execute_reply":"2022-06-03T06:36:47.783375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from autogluon.tabular import TabularDataset, TabularPredictor\nimport autogluon.core as ag","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:47.788484Z","iopub.execute_input":"2022-06-03T06:36:47.788885Z","iopub.status.idle":"2022-06-03T06:36:49.92295Z","shell.execute_reply.started":"2022-06-03T06:36:47.788849Z","shell.execute_reply":"2022-06-03T06:36:49.921579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = train_labels.sort_values(by=['customer_ID']).drop('customer_ID', axis=1)\ntrain_df = pd.concat([train_df, train_labels], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:49.925698Z","iopub.execute_input":"2022-06-03T06:36:49.926217Z","iopub.status.idle":"2022-06-03T06:36:51.677867Z","shell.execute_reply.started":"2022-06-03T06:36:49.926168Z","shell.execute_reply":"2022-06-03T06:36:51.676646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:51.679657Z","iopub.execute_input":"2022-06-03T06:36:51.680168Z","iopub.status.idle":"2022-06-03T06:36:51.879031Z","shell.execute_reply.started":"2022-06-03T06:36:51.680119Z","shell.execute_reply":"2022-06-03T06:36:51.877819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:51.881247Z","iopub.execute_input":"2022-06-03T06:36:51.881754Z","iopub.status.idle":"2022-06-03T06:36:51.895816Z","shell.execute_reply.started":"2022-06-03T06:36:51.881705Z","shell.execute_reply":"2022-06-03T06:36:51.894952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from autogluon.core.metrics import make_scorer\namex_scorer = make_scorer(name='amex_metric',\n                          score_func=amex_metric_mod,\n                          optimum=1,\n                           greater_is_better=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:51.897357Z","iopub.execute_input":"2022-06-03T06:36:51.897756Z","iopub.status.idle":"2022-06-03T06:36:51.914702Z","shell.execute_reply.started":"2022-06-03T06:36:51.897722Z","shell.execute_reply":"2022-06-03T06:36:51.913202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nhyperparameters = {  # hyperparameters of each model type\n                   'GBM': {},\n                   'NN_TORCH': {},  # NOTE: comment this line out if you get errors on Mac OSX\n                  }","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:51.91664Z","iopub.execute_input":"2022-06-03T06:36:51.917577Z","iopub.status.idle":"2022-06-03T06:36:51.931111Z","shell.execute_reply.started":"2022-06-03T06:36:51.917523Z","shell.execute_reply":"2022-06-03T06:36:51.92997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:51.932592Z","iopub.execute_input":"2022-06-03T06:36:51.932931Z","iopub.status.idle":"2022-06-03T06:36:52.099334Z","shell.execute_reply.started":"2022-06-03T06:36:51.9329Z","shell.execute_reply":"2022-06-03T06:36:52.098326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:52.101878Z","iopub.execute_input":"2022-06-03T06:36:52.102307Z","iopub.status.idle":"2022-06-03T06:36:52.116294Z","shell.execute_reply.started":"2022-06-03T06:36:52.102266Z","shell.execute_reply":"2022-06-03T06:36:52.114996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in range(5):\n#     if i != 4:\n#         temp_df = train_df.sample(n=91782, random_state=42)\n#     else:\n#         temp_df = train_df\n#     print(temp_df.shape)\n#     print(temp_df['target'].value_counts())\n#     train_df = train_df.drop(temp_df.index, axis=0)\n#     # break","metadata":{"execution":{"iopub.status.busy":"2022-06-03T06:36:52.118367Z","iopub.execute_input":"2022-06-03T06:36:52.118886Z","iopub.status.idle":"2022-06-03T06:36:52.128853Z","shell.execute_reply.started":"2022-06-03T06:36:52.118841Z","shell.execute_reply":"2022-06-03T06:36:52.127742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_path = 'agModels-predictClass'  # specifies folder to store trained models\npredictor = TabularPredictor(label='target', path=save_path, eval_metric='recall').fit(train_df.drop('customer_ID', axis=1), \n                                                                                          hyperparameters=hyperparameters, \n                                                                                          auto_stack=True,\n                                                                                           num_stack_levels=0\n                                                                                          # time_limit=10*60,\n                                                                                          #ag_args_fit={'num_gpus': 1}\n                                                                                      )","metadata":{"execution":{"iopub.status.busy":"2022-06-03T12:26:43.960856Z","iopub.execute_input":"2022-06-03T12:26:43.961383Z","iopub.status.idle":"2022-06-03T12:28:37.089983Z","shell.execute_reply.started":"2022-06-03T12:26:43.961341Z","shell.execute_reply":"2022-06-03T12:28:37.088522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}