{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"ref\n\n1. https://www.kaggle.com/code/junjitakeshima/amex-simple-lgbm-starter-for-beginner-en  \n2. https://www.kaggle.com/code/junjitakeshima/amex-try-to-improve-lgbm-starter-eng","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport lightgbm as lgbm\nfrom lightgbm import LGBMClassifier\npd.set_option(\"display.max_columns\", None)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:57:45.093954Z","iopub.execute_input":"2022-07-05T03:57:45.095567Z","iopub.status.idle":"2022-07-05T03:57:45.104277Z","shell.execute_reply.started":"2022-07-05T03:57:45.095503Z","shell.execute_reply":"2022-07-05T03:57:45.103149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# from ref1 [Public Score = 0.783]","metadata":{}},{"cell_type":"markdown","source":"Simple LGBM","metadata":{}},{"cell_type":"code","source":"\ntrain_df = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet').groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()\ntrain_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv').set_index('customer_ID', drop=True).sort_index()\ntrain_df = pd.merge(train_df, train_labels, left_index=True, right_index=True)\n\nall_cols = train_df.columns\nnon_use_cols = ['S_2','B_30','B_38','D_114','D_116','D_117','D_120','D_126','D_63','D_64','D_66','D_68', 'target']\nfeature_cols = [col for col in all_cols if col not in non_use_cols]\n\ny = train_df['target'].copy()\nx = train_df[feature_cols]\n\nmodel_lgbm = lgbm.LGBMClassifier(n_estimators = 300)\ndel train_df\ndel train_labels\nmodel_lgbm.fit(x, y)\n\ntest_df = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet').groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()\ntest_df = test_df[feature_cols]\ny_pred  = model_lgbm.predict_proba(test_df)\n\nsub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsub[\"prediction\"] = y_pred[:,1]\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:57:45.113412Z","iopub.execute_input":"2022-07-05T03:57:45.113740Z","iopub.status.idle":"2022-07-05T03:59:24.367390Z","shell.execute_reply.started":"2022-07-05T03:57:45.113712Z","shell.execute_reply":"2022-07-05T03:59:24.366467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# from ref2 [Public Score = 0.787]","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## (2) Increase Train data [Public Score = 0.784]\n\n","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet').groupby('customer_ID').tail(2).set_index('customer_ID', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:59:24.373481Z","iopub.execute_input":"2022-07-05T03:59:24.375503Z","iopub.status.idle":"2022-07-05T03:59:45.075396Z","shell.execute_reply.started":"2022-07-05T03:59:24.375450Z","shell.execute_reply":"2022-07-05T03:59:45.074237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv').set_index('customer_ID', drop=True).sort_index()\ntrain_df = pd.merge(train_df, train_labels, left_index=True, right_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:59:45.076913Z","iopub.execute_input":"2022-07-05T03:59:45.077241Z","iopub.status.idle":"2022-07-05T03:59:59.328254Z","shell.execute_reply.started":"2022-07-05T03:59:45.077209Z","shell.execute_reply":"2022-07-05T03:59:59.327128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## (3) Removing Outliers [Public Score = 0.785]","metadata":{}},{"cell_type":"code","source":"all_cols = train_df.columns\nnon_use_cols = ['S_2','B_30','B_38','D_114','D_116','D_117','D_120','D_126','D_63','D_64','D_66','D_68', 'target']\nfeature_cols = [col for col in all_cols if col not in non_use_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:59:59.332121Z","iopub.execute_input":"2022-07-05T03:59:59.333068Z","iopub.status.idle":"2022-07-05T03:59:59.338856Z","shell.execute_reply.started":"2022-07-05T03:59:59.333029Z","shell.execute_reply":"2022-07-05T03:59:59.337444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outlier_list = []\noutlier_col = []\n\nfor col in feature_cols :\n    \n    temp_df = train_df[(train_df[col] > train_df[col].mean() + train_df[col].std() * 200) |\n                       (train_df[col] < train_df[col].mean() - train_df[col].std() * 200) ]\n    if len(temp_df) >0 and len(temp_df) <6 : \n        outliers = temp_df.index.to_list()\n        outlier_list.extend(outliers)\n        outlier_col.append(col)\n        print(col, len(temp_df))\n    \noutlier_list = list(set(outlier_list))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:59:59.340197Z","iopub.execute_input":"2022-07-05T03:59:59.340595Z","iopub.status.idle":"2022-07-05T04:00:03.208690Z","shell.execute_reply.started":"2022-07-05T03:59:59.340536Z","shell.execute_reply":"2022-07-05T04:00:03.207475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(outlier_list, inplace = True)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:00:03.210244Z","iopub.execute_input":"2022-07-05T04:00:03.210571Z","iopub.status.idle":"2022-07-05T04:00:03.995127Z","shell.execute_reply.started":"2022-07-05T04:00:03.210540Z","shell.execute_reply":"2022-07-05T04:00:03.993951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## (4) Cross validation (KFold = 3) [Public Score = 0.786]\n## (5) Tuning parameters [Public Score = 0.787]","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score\n\nkf = KFold(n_splits = 3)\nmodels = []\nlgbm_params ={\"objective\":\"binary\",\n              'num_leaves': 69,\n              'max_bin': 312,\n              'feature_fraction': 0.5193052325515839,\n              'bagging_fraction': 0.597903764208054,\n              'bagging_freq': 0,\n              'min_data_in_leaf': 21,\n              \"random_seed\":1234}\n\nfor train_index, val_index in kf.split(x):\n    X_train = x.iloc[train_index]\n    X_valid = x.iloc[val_index]\n    Y_train = y.iloc[train_index]\n    Y_valid = y.iloc[val_index]\n    \n    lgbm_train = lgbm.Dataset(X_train, Y_train)\n    lgbm_eval = lgbm.Dataset(X_valid, Y_valid, reference=lgbm_train)\n    \n    model_lgbm = lgbm.train(lgbm_params,\n                           lgbm_train,\n                           valid_sets = lgbm_eval,\n                           num_boost_round = 300,\n                           early_stopping_rounds = 20,\n                           verbose_eval = 10,\n                           )\n    y_pred = model_lgbm.predict(X_valid, num_iteration = model_lgbm.best_iteration)\n        \n    print (accuracy_score(Y_valid, np.round(y_pred)))\n    \n    models.append(model_lgbm)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:00:03.996409Z","iopub.execute_input":"2022-07-05T04:00:03.996826Z","iopub.status.idle":"2022-07-05T04:01:29.036035Z","shell.execute_reply.started":"2022-07-05T04:00:03.996767Z","shell.execute_reply":"2022-07-05T04:01:29.035261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ndel train_labels\ndel x","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:01:29.037144Z","iopub.execute_input":"2022-07-05T04:01:29.037913Z","iopub.status.idle":"2022-07-05T04:01:29.061610Z","shell.execute_reply.started":"2022-07-05T04:01:29.037879Z","shell.execute_reply":"2022-07-05T04:01:29.060144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_df = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet').groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()\ntest_df = test_df[feature_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:01:29.063229Z","iopub.execute_input":"2022-07-05T04:01:29.063676Z","iopub.status.idle":"2022-07-05T04:01:56.368499Z","shell.execute_reply.started":"2022-07-05T04:01:29.063631Z","shell.execute_reply":"2022-07-05T04:01:56.367269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\n\nfor model in models:\n    pred = model.predict(test_df)\n    preds.append(pred)\n    \npreds_array = np.array(preds)\npreds_mean = np.mean(preds_array, axis =0)\n\npreds_mean","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:01:56.374033Z","iopub.execute_input":"2022-07-05T04:01:56.374571Z","iopub.status.idle":"2022-07-05T04:02:14.160852Z","shell.execute_reply.started":"2022-07-05T04:01:56.374536Z","shell.execute_reply":"2022-07-05T04:02:14.159619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsub[\"prediction\"] = preds_mean\nsub.to_csv('submission.csv', index=False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T04:02:14.162392Z","iopub.execute_input":"2022-07-05T04:02:14.163322Z","iopub.status.idle":"2022-07-05T04:02:19.869489Z","shell.execute_reply.started":"2022-07-05T04:02:14.163281Z","shell.execute_reply":"2022-07-05T04:02:19.868396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}