{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom collections import Counter\nfrom sklearn.model_selection import KFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-02T10:20:43.445695Z","iopub.execute_input":"2022-07-02T10:20:43.446125Z","iopub.status.idle":"2022-07-02T10:20:43.456169Z","shell.execute_reply.started":"2022-07-02T10:20:43.446080Z","shell.execute_reply":"2022-07-02T10:20:43.455067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mode(x):\n    return x.mode().iloc[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-02T10:27:00.878658Z","iopub.execute_input":"2022-07-02T10:27:00.879064Z","iopub.status.idle":"2022-07-02T10:27:00.884049Z","shell.execute_reply.started":"2022-07-02T10:27:00.879031Z","shell.execute_reply":"2022-07-02T10:27:00.883108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_parquet(\"../input/amex-parquet/train_data.parquet\")\ndrop_cols = [\"S_2\", \"D_87\", \"D_88\", \"D_108\", \"D_110\", \"D_111\", \"B_39\", \"D_73\", \"B_42\", \"D_66\", \"D_134\", \"D_135\", \"D_136\", \"D_137\", \"D_138\", \"R_9\"]\ndf_train.drop(drop_cols, axis=1, inplace=True)\n\ncat_features = [\n    \"B_30\", \"B_38\",\n    \"D_63\", \"D_64\", \"D_68\",\n    \"D_114\", \"D_116\", \"D_117\", \"D_120\", \"D_126\",\n]\n\ntrain_y = df_train[[\"customer_ID\", \"target\"]].groupby(\"customer_ID\")[\"target\"].last()\ndf_train.drop(\"target\", axis=1, inplace=True)\n\n# See categorical features value_counts\n# for col in cat_features:\n#     print(col, df_train[col].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-02T10:20:43.465367Z","iopub.execute_input":"2022-07-02T10:20:43.465944Z","iopub.status.idle":"2022-07-02T10:21:30.898597Z","shell.execute_reply.started":"2022-07-02T10:20:43.465899Z","shell.execute_reply":"2022-07-02T10:21:30.897163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numerical features\nnum_features = [col for col in df_train.columns if col not in cat_features and col != \"customer_ID\"]\ndf_train_num = df_train.groupby(\"customer_ID\")[num_features].agg(['mean', 'min', 'max', 'last'])\ndf_train.drop(num_features, axis=1, inplace=True)\n\ndf_train = df_train.fillna(-999)\n\n# categorical features\ndf_train_cat = df_train.groupby(\"customer_ID\")[cat_features].agg([mode, 'last'])\n\n# df_train0 = df_train.query(\"target == 0\").sample(n=1377869, random_state=0)\n# df_train1 = df_train.query(\"target == 1\")\n# del df_train; gc.collect()\n# df_train = pd.concat([df_train0, df_train1])\n# del df_train0, df_train1; gc.collect()\n# df_train.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T10:31:09.681086Z","iopub.execute_input":"2022-07-02T10:31:09.681482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.concat([df_train_num, df_train_cat], axis=1)\n\ndel df_train_num, df_train_cat\ngc.collect()\n\ndf_train.columns = [col[0] + \"_\" + col[1] for col in df_train.columns]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = df_train\ndel df_train; gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    'boosting_type': 'gbdt',  # default = 'gbdt'\n    'num_leaves': 63,         # default = 31,\n    'learning_rate': 0.1,     # default = 0.1\n    'feature_fraction': 0.8,  # default = 1.0\n    'bagging_freq': 1,        # default = 0\n    'bagging_fraction': 0.8,  # default = 1.0\n    'n_estimators': 10000,\n    'random_state': 0,        # default = None\n}\n\n# Thanks!\n# https://www.kaggle.com/code/munumbutt/simple-lgbm-starter\ncat_cols = []\nfor col in cat_features:\n    cat_cols.append(f\"{col}_mode\")\n    cat_cols.append(f\"{col}_last\")\n    train_x[f\"{col}_mode\"] = train_x[f\"{col}_mode\"].astype(\"category\")\n    train_x[f\"{col}_last\"] = train_x[f\"{col}_last\"].astype(\"category\")\n\ncv = KFold(n_splits=5)\nfor fold, (trn_idx, val_idx) in enumerate(cv.split(train_x), start=1):\n    # def main():\n    trn_x, trn_y = train_x.iloc[trn_idx, :], train_y[trn_idx]\n    val_x, val_y = train_x.iloc[val_idx, :], train_y[val_idx]\n    print(trn_x.shape, trn_y.shape, val_x.shape, val_y.shape)\n    clf = lgb.LGBMClassifier(**params)\n    clf.fit(\n        trn_x, trn_y, \n        eval_set=[(val_x, val_y)],\n        callbacks=[lgb.early_stopping(50), lgb.log_evaluation(200)],\n        categorical_feature=cat_cols\n    )\n    del trn_x, trn_y, val_x, val_y; gc.collect()\n    pickle.dump(clf, open(f\"model.lgb.{fold}.pkl\", 'wb'))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Infer","metadata":{}},{"cell_type":"code","source":"# train_x_cols = train_x.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:34:57.649128Z","iopub.execute_input":"2022-06-21T15:34:57.649467Z","iopub.status.idle":"2022-06-21T15:34:57.653958Z","shell.execute_reply.started":"2022-06-21T15:34:57.649438Z","shell.execute_reply":"2022-06-21T15:34:57.652807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train_x; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:35:00.567619Z","iopub.execute_input":"2022-06-21T15:35:00.56804Z","iopub.status.idle":"2022-06-21T15:35:00.894341Z","shell.execute_reply.started":"2022-06-21T15:35:00.568007Z","shell.execute_reply":"2022-06-21T15:35:00.893587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols_trn.append(\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-06-21T14:57:51.665022Z","iopub.execute_input":"2022-06-21T14:57:51.665624Z","iopub.status.idle":"2022-06-21T14:57:51.670426Z","shell.execute_reply.started":"2022-06-21T14:57:51.665577Z","shell.execute_reply":"2022-06-21T14:57:51.669668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clfs = []\n# for fold in [1, 2, 3, 4, 5]:\n#     clfs.append(pickle.load(open(f\"../input/amex-1st-lgb/model.lgb.{fold}.pkl\", \"rb\")))","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:57:07.942048Z","iopub.execute_input":"2022-06-21T15:57:07.942477Z","iopub.status.idle":"2022-06-21T15:57:08.868144Z","shell.execute_reply.started":"2022-06-21T15:57:07.942446Z","shell.execute_reply":"2022-06-21T15:57:08.867031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(cols_trn)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:50:48.071772Z","iopub.execute_input":"2022-06-21T15:50:48.072215Z","iopub.status.idle":"2022-06-21T15:50:48.07886Z","shell.execute_reply.started":"2022-06-21T15:50:48.072179Z","shell.execute_reply":"2022-06-21T15:50:48.077832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train_head = pd.read_pickle(\"df_train.agg.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:57:13.246688Z","iopub.execute_input":"2022-06-21T15:57:13.247217Z","iopub.status.idle":"2022-06-21T15:57:13.262627Z","shell.execute_reply.started":"2022-06-21T15:57:13.247172Z","shell.execute_reply":"2022-06-21T15:57:13.261492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(set([col[0] for col in df_train_head.columns]))","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:58:22.908867Z","iopub.execute_input":"2022-06-21T15:58:22.909707Z","iopub.status.idle":"2022-06-21T15:58:22.92Z","shell.execute_reply.started":"2022-06-21T15:58:22.909656Z","shell.execute_reply":"2022-06-21T15:58:22.919045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import collections","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:58:43.670114Z","iopub.execute_input":"2022-06-21T15:58:43.670568Z","iopub.status.idle":"2022-06-21T15:58:43.675566Z","shell.execute_reply.started":"2022-06-21T15:58:43.67053Z","shell.execute_reply":"2022-06-21T15:58:43.67455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = sorted(set([col[0] for col in df_train_head.columns]))\n# cols.append(\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-06-21T16:01:15.594784Z","iopub.execute_input":"2022-06-21T16:01:15.595913Z","iopub.status.idle":"2022-06-21T16:01:15.601234Z","shell.execute_reply.started":"2022-06-21T16:01:15.595853Z","shell.execute_reply":"2022-06-21T16:01:15.600396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"HOGE\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test = pd.read_feather(\"../input/amexfeather/test_data.ftr\", columns=cols).groupby(\"customer_ID\").agg(['mean', 'std', 'min', 'max', 'last'])\n# drop_cols = [\"S_2\", \"D_87\", \"D_88\", \"D_108\", \"D_110\", \"D_111\", \"B_39\", \"D_73\", \"B_42\", \"D_134\", \"D_135\", \"D_136\", \"D_137\", \"D_138\", \"R_9\"]\n# df_test.drop(drop_cols, axis=1, inplace=True)\n# for col in [\"D_63\", \"D_64\"]:\n#     d = {}\n#     for i, val in enumerate(df_test[col].value_counts().keys()):\n#         d[val] = i\n#     df_test[col] = df_test[col].replace(d).astype(\"category\")\n# preds_y = np.zeros(len(df_test))\n# for fold in range(5):\n#     preds_y += clfs[fold].predict_proba(df_test.values)[:, 1] / 5.0","metadata":{"execution":{"iopub.status.busy":"2022-06-21T16:01:21.141533Z","iopub.execute_input":"2022-06-21T16:01:21.142086Z","iopub.status.idle":"2022-06-21T16:04:29.183436Z","shell.execute_reply.started":"2022-06-21T16:01:21.142042Z","shell.execute_reply":"2022-06-21T16:04:29.173655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(cols_trn)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T15:44:02.184427Z","iopub.execute_input":"2022-06-21T15:44:02.185382Z","iopub.status.idle":"2022-06-21T15:44:02.190787Z","shell.execute_reply.started":"2022-06-21T15:44:02.185335Z","shell.execute_reply":"2022-06-21T15:44:02.190142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test","metadata":{"execution":{"iopub.status.busy":"2022-06-21T16:07:34.309123Z","iopub.execute_input":"2022-06-21T16:07:34.309686Z","iopub.status.idle":"2022-06-21T16:07:34.996026Z","shell.execute_reply.started":"2022-06-21T16:07:34.309644Z","shell.execute_reply":"2022-06-21T16:07:34.994868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set(cols) - set([col[0] for col in df_test.columns])","metadata":{"execution":{"iopub.status.busy":"2022-06-21T16:08:01.305088Z","iopub.execute_input":"2022-06-21T16:08:01.30586Z","iopub.status.idle":"2022-06-21T16:08:01.31355Z","shell.execute_reply.started":"2022-06-21T16:08:01.305821Z","shell.execute_reply":"2022-06-21T16:08:01.31246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Make submission files\n# df_sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\n# df_sub[\"prediction\"] = preds_y\n# df_sub.to_csv(\"submission.csv.gz\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T11:04:48.690074Z","iopub.execute_input":"2022-05-27T11:04:48.690732Z","iopub.status.idle":"2022-05-27T11:04:50.37935Z","shell.execute_reply.started":"2022-05-27T11:04:48.690667Z","shell.execute_reply":"2022-05-27T11:04:50.377907Z"},"trusted":true},"execution_count":null,"outputs":[]}]}