{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AutoGluon AutoML for the *American Express - Default Prediction* competition ","metadata":{}},{"cell_type":"code","source":"!pip install autogluon.tabular[all]  # Only install autogluon.tabular module and all its dependencies","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-29T00:38:36.588931Z","iopub.execute_input":"2022-06-29T00:38:36.589475Z","iopub.status.idle":"2022-06-29T00:39:04.13974Z","shell.execute_reply.started":"2022-06-29T00:38:36.589361Z","shell.execute_reply":"2022-06-29T00:39:04.138031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from autogluon.tabular import TabularPredictor, TabularDataset\n\nimport pandas as pd\nimport numpy as np\nfrom IPython.display import display\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-06-29T00:39:04.142627Z","iopub.execute_input":"2022-06-29T00:39:04.143086Z","iopub.status.idle":"2022-06-29T00:39:05.377813Z","shell.execute_reply.started":"2022-06-29T00:39:04.143048Z","shell.execute_reply":"2022-06-29T00:39:05.376559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do some basic data preprocessing following [this notebook by @ambrosm](https://www.kaggle.com/code/ambrosm/amex-lightgbm-quickstart) ","metadata":{}},{"cell_type":"code","source":"features_avg = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_20', 'B_21', 'B_22', 'B_23', 'B_24', 'B_25', 'B_28', 'B_29', 'B_30', 'B_32', 'B_33', 'B_37', 'B_38', 'B_39', 'B_40', 'B_41', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_50', 'D_51', 'D_53', 'D_54', 'D_55', 'D_58', 'D_59', 'D_60', 'D_61', 'D_62', 'D_65', 'D_66', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_74', 'D_75', 'D_76', 'D_77', 'D_78', 'D_80', 'D_82', 'D_84', 'D_86', 'D_91', 'D_92', 'D_94', 'D_96', 'D_103', 'D_104', 'D_108', 'D_112', 'D_113', 'D_114', 'D_115', 'D_117', 'D_118', 'D_119', 'D_120', 'D_121', 'D_122', 'D_123', 'D_124', 'D_125', 'D_126', 'D_128', 'D_129', 'D_131', 'D_132', 'D_133', 'D_134', 'D_135', 'D_136', 'D_140', 'D_141', 'D_142', 'D_144', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_2', 'R_3', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_14', 'R_15', 'R_16', 'R_17', 'R_20', 'R_21', 'R_22', 'R_24', 'R_26', 'R_27', 'S_3', 'S_5', 'S_6', 'S_7', 'S_9', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_18', 'S_22', 'S_23', 'S_25', 'S_26']\nfeatures_min = ['B_2', 'B_4', 'B_5', 'B_9', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_19', 'B_20', 'B_28', 'B_29', 'B_33', 'B_36', 'B_42', 'D_39', 'D_41', 'D_42', 'D_45', 'D_46', 'D_48', 'D_50', 'D_51', 'D_53', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_62', 'D_70', 'D_71', 'D_74', 'D_75', 'D_78', 'D_83', 'D_102', 'D_112', 'D_113', 'D_115', 'D_118', 'D_119', 'D_121', 'D_122', 'D_128', 'D_132', 'D_140', 'D_141', 'D_144', 'D_145', 'P_2', 'P_3', 'R_1', 'R_27', 'S_3', 'S_5', 'S_7', 'S_9', 'S_11', 'S_12', 'S_23', 'S_25']\nfeatures_max = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_21', 'B_23', 'B_24', 'B_25', 'B_29', 'B_30', 'B_33', 'B_37', 'B_38', 'B_39', 'B_40', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_52', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_61', 'D_63', 'D_64', 'D_65', 'D_70', 'D_71', 'D_72', 'D_73', 'D_74', 'D_76', 'D_77', 'D_78', 'D_80', 'D_82', 'D_84', 'D_91', 'D_102', 'D_105', 'D_107', 'D_110', 'D_111', 'D_112', 'D_115', 'D_116', 'D_117', 'D_118', 'D_119', 'D_121', 'D_122', 'D_123', 'D_124', 'D_125', 'D_126', 'D_128', 'D_131', 'D_132', 'D_133', 'D_134', 'D_135', 'D_136', 'D_138', 'D_140', 'D_141', 'D_142', 'D_144', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_3', 'R_5', 'R_6', 'R_7', 'R_8', 'R_10', 'R_11', 'R_14', 'R_17', 'R_20', 'R_26', 'R_27', 'S_3', 'S_5', 'S_7', 'S_8', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\nfeatures_last = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_20', 'B_21', 'B_22', 'B_23', 'B_24', 'B_25', 'B_26', 'B_28', 'B_29', 'B_30', 'B_32', 'B_33', 'B_36', 'B_37', 'B_38', 'B_39', 'B_40', 'B_41', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_51', 'D_52', 'D_53', 'D_54', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_61', 'D_62', 'D_63', 'D_64', 'D_65', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_75', 'D_76', 'D_77', 'D_78', 'D_79', 'D_80', 'D_81', 'D_82', 'D_83', 'D_86', 'D_91', 'D_96', 'D_105', 'D_106', 'D_112', 'D_114', 'D_119', 'D_120', 'D_121', 'D_122', 'D_124', 'D_125', 'D_126', 'D_127', 'D_130', 'D_131', 'D_132', 'D_133', 'D_134', 'D_138', 'D_140', 'D_141', 'D_142', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_2', 'R_3', 'R_4', 'R_5', 'R_6', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_12', 'R_13', 'R_14', 'R_15', 'R_19', 'R_20', 'R_26', 'R_27', 'S_3', 'S_5', 'S_6', 'S_7', 'S_8', 'S_9', 'S_11', 'S_12', 'S_13', 'S_16', 'S_19', 'S_20', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\nfor i in ['test', 'train']:\n    df = pd.read_parquet(f'../input/amex-data-integer-dtypes-parquet-format/{i}.parquet')\n    cid = pd.Categorical(df.pop('customer_ID'), ordered=True)\n    last = (cid != np.roll(cid, -1)) # mask for last statement of every customer\n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n    df_avg = (df\n              .groupby(cid)\n              .mean()[features_avg]\n              .rename(columns={f: f\"{f}_avg\" for f in features_avg})\n             )\n    gc.collect()\n    print('Computed avg', i)\n    df_min = (df\n              .groupby(cid)\n              .min()[features_min]\n              .rename(columns={f: f\"{f}_min\" for f in features_min})\n             )\n    gc.collect()\n    print('Computed min', i)\n    df_max = (df\n              .groupby(cid)\n              .max()[features_max]\n              .rename(columns={f: f\"{f}_max\" for f in features_max})\n             )\n    gc.collect()\n    print('Computed max', i)\n    df = (df.loc[last, features_last]\n          .rename(columns={f: f\"{f}_last\" for f in features_last})\n          .set_index(np.asarray(cid[last]))\n         )\n    gc.collect()\n    print('Computed last', i)\n    df = pd.concat([df, df_min, df_max, df_avg], axis=1)\n    if i == 'train': train = df\n    else: test = df\n    print(f\"{i} shape: {df.shape}\")\n    del df, df_avg, df_min, df_max, cid, last\n\ntarget = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-06-29T00:39:05.379505Z","iopub.execute_input":"2022-06-29T00:39:05.379904Z","iopub.status.idle":"2022-06-29T00:45:42.068015Z","shell.execute_reply.started":"2022-06-29T00:39:05.379871Z","shell.execute_reply":"2022-06-29T00:45:42.06665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Format data:\nfeatures = [f for f in test.columns if f != 'customer_ID' and f != 'target']\ntrain = train[features]\ntrain[\"label\"] = target\ntrain = TabularDataset(train)\ntest = TabularDataset(test[features])\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-29T00:45:42.069431Z","iopub.execute_input":"2022-06-29T00:45:42.070462Z","iopub.status.idle":"2022-06-29T00:45:45.107107Z","shell.execute_reply.started":"2022-06-29T00:45:42.070425Z","shell.execute_reply":"2022-06-29T00:45:45.105842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train model","metadata":{}},{"cell_type":"code","source":"NUM_HOURS_TO_TRAIN = 2.0  # you should get better performance if you increase this\neval_metric = \"roc_auc\"  # metric to optimize on holdout data, you can play with different options or define custom metric\npath = \"ag_model\"  # file in which to save trained model\n\npredictor = TabularPredictor(label=\"label\", eval_metric=eval_metric, path=path)\npredictor.fit(train, time_limit=NUM_HOURS_TO_TRAIN*60*60, presets=[\"best_quality\",\"optimize_for_deployment\"])\n\npredictor.leaderboard()  # Look at validation score for each individual model trained by the AutoML system","metadata":{"execution":{"iopub.status.busy":"2022-06-29T00:45:45.110214Z","iopub.execute_input":"2022-06-29T00:45:45.111208Z","iopub.status.idle":"2022-06-29T01:45:51.1536Z","shell.execute_reply.started":"2022-06-29T00:45:45.111171Z","shell.execute_reply":"2022-06-29T01:45:51.152364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Generate test data predictions","metadata":{}},{"cell_type":"code","source":"pred_test = predictor.predict_proba(test, as_pandas=False, as_multiclass=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T01:45:51.155362Z","iopub.execute_input":"2022-06-29T01:45:51.155737Z","iopub.status.idle":"2022-06-29T01:49:53.220626Z","shell.execute_reply.started":"2022-06-29T01:45:51.155705Z","shell.execute_reply":"2022-06-29T01:49:53.219347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Generate submission","metadata":{}},{"cell_type":"code","source":"sub = pd.DataFrame({'customer_ID': test.index, 'prediction': pred_test})\nsub.to_csv('submission.csv', index=False)\ndisplay(sub)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T01:49:53.222463Z","iopub.execute_input":"2022-06-29T01:49:53.223485Z","iopub.status.idle":"2022-06-29T01:49:55.957453Z","shell.execute_reply.started":"2022-06-29T01:49:53.223442Z","shell.execute_reply":"2022-06-29T01:49:55.956265Z"},"trusted":true},"execution_count":null,"outputs":[]}]}