{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-06-22T07:48:55.826284Z","iopub.execute_input":"2022-06-22T07:48:55.826735Z","iopub.status.idle":"2022-06-22T07:48:55.860807Z","shell.execute_reply.started":"2022-06-22T07:48:55.826647Z","shell.execute_reply":"2022-06-22T07:48:55.859906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport lightgbm as lgbm\nfrom lightgbm import LGBMClassifier\npd.set_option(\"display.max_columns\", None)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T07:49:03.414612Z","iopub.execute_input":"2022-06-22T07:49:03.415027Z","iopub.status.idle":"2022-06-22T07:49:04.973556Z","shell.execute_reply.started":"2022-06-22T07:49:03.414988Z","shell.execute_reply":"2022-06-22T07:49:04.972494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reading the whole data would cause memory-error, so we'll read two(2) records per each customer(tail(2)), like in [this](https://www.kaggle.com/code/junjitakeshima/amex-simple-lgbm-starter-for-beginner-en) notebook.","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet').groupby('customer_ID').tail(2).set_index('customer_ID', drop=True).sort_index()\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:04:36.901838Z","iopub.execute_input":"2022-06-22T08:04:36.90229Z","iopub.status.idle":"2022-06-22T08:05:13.496582Z","shell.execute_reply.started":"2022-06-22T08:04:36.902236Z","shell.execute_reply":"2022-06-22T08:05:13.494801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv').set_index('customer_ID', drop=True).sort_index()\ntrain_labels","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:05:38.229589Z","iopub.execute_input":"2022-06-22T08:05:38.230143Z","iopub.status.idle":"2022-06-22T08:05:39.339534Z","shell.execute_reply.started":"2022-06-22T08:05:38.230098Z","shell.execute_reply":"2022-06-22T08:05:39.338315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('number of default:', train_labels['target'].sum())\nprint('percent of default:', train_labels['target'].sum() / train_labels['target'].count() * 100, '%')","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:08:45.440257Z","iopub.execute_input":"2022-06-22T08:08:45.440738Z","iopub.status.idle":"2022-06-22T08:08:45.449112Z","shell.execute_reply.started":"2022-06-22T08:08:45.440703Z","shell.execute_reply":"2022-06-22T08:08:45.448257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.merge(train_df, train_labels, left_index=True, right_index=True)\ntrain_df.target = train_df.target.astype('int8')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:13:48.62903Z","iopub.execute_input":"2022-06-22T08:13:48.629428Z","iopub.status.idle":"2022-06-22T08:13:48.80048Z","shell.execute_reply.started":"2022-06-22T08:13:48.629394Z","shell.execute_reply":"2022-06-22T08:13:48.799116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train_df.columns[1:-1]\nprint(f'There are {len(features)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:14:41.477256Z","iopub.execute_input":"2022-06-22T08:14:41.478792Z","iopub.status.idle":"2022-06-22T08:14:41.487094Z","shell.execute_reply.started":"2022-06-22T08:14:41.478742Z","shell.execute_reply":"2022-06-22T08:14:41.485847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = train_df.columns\nnon_use_cols = ['S_2','B_30','B_38','D_114','D_116','D_117','D_120','D_126','D_63','D_64','D_66','D_68', 'target']\nfeature_cols = [col for col in all_cols if col not in non_use_cols]","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:15:47.673544Z","iopub.execute_input":"2022-06-22T08:15:47.674098Z","iopub.status.idle":"2022-06-22T08:15:47.68212Z","shell.execute_reply.started":"2022-06-22T08:15:47.674037Z","shell.execute_reply":"2022-06-22T08:15:47.680933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(train_df[feature_cols], train_df['target'].copy(),\n                                                    stratify=train_df['target'].copy(), \n                                                    test_size=0.20)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:50:38.35986Z","iopub.execute_input":"2022-06-22T08:50:38.361725Z","iopub.status.idle":"2022-06-22T08:50:40.857829Z","shell.execute_reply.started":"2022-06-22T08:50:38.361667Z","shell.execute_reply":"2022-06-22T08:50:40.856461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lgbm = lgbm.LGBMClassifier(n_estimators = 300)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:50:54.28041Z","iopub.execute_input":"2022-06-22T08:50:54.28085Z","iopub.status.idle":"2022-06-22T08:50:54.287515Z","shell.execute_reply.started":"2022-06-22T08:50:54.280811Z","shell.execute_reply":"2022-06-22T08:50:54.286083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_params={\"early_stopping_rounds\":30, \n            \"eval_metric\" : 'auc', \n            \"eval_set\" : [(X_test_,y_test_)],\n            'eval_names': ['valid'],\n            'verbose': 100}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel_lgbm.fit(X_train, y_train, eval_set=[(X_val, y_val),])","metadata":{"execution":{"iopub.status.busy":"2022-06-22T08:50:56.983414Z","iopub.execute_input":"2022-06-22T08:50:56.983801Z","iopub.status.idle":"2022-06-22T08:51:58.059591Z","shell.execute_reply.started":"2022-06-22T08:50:56.983768Z","shell.execute_reply":"2022-06-22T08:51:58.058484Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nvalues = model_lgbm.booster_.feature_importance()\nkeys = model_lgbm.booster_.feature_name()\n\ndata = pd.DataFrame(data=values, index=keys, columns=[\"score\"]).sort_values(by = \"score\", ascending=False)\ndata.nlargest(25, columns=\"score\").plot(kind='barh', figsize = (20,10)) ## plot top 40 features","metadata":{"execution":{"iopub.status.busy":"2022-06-22T09:04:33.758373Z","iopub.execute_input":"2022-06-22T09:04:33.758801Z","iopub.status.idle":"2022-06-22T09:04:34.678831Z","shell.execute_reply.started":"2022-06-22T09:04:33.758767Z","shell.execute_reply":"2022-06-22T09:04:34.6777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T09:05:33.615397Z","iopub.execute_input":"2022-06-22T09:05:33.616303Z","iopub.status.idle":"2022-06-22T09:05:33.629082Z","shell.execute_reply.started":"2022-06-22T09:05:33.616235Z","shell.execute_reply":"2022-06-22T09:05:33.6281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = amex_metric_mod(y_valid.values, oof_preds)\nprint('Kaggle Metric =',acc,'\\n')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_df = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet').groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-06-22T09:10:30.97034Z","iopub.execute_input":"2022-06-22T09:10:30.970756Z","iopub.status.idle":"2022-06-22T09:11:25.699516Z","shell.execute_reply.started":"2022-06-22T09:10:30.970722Z","shell.execute_reply":"2022-06-22T09:11:25.698352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df[feature_cols]\ny_pred  = model_lgbm.predict_proba(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T09:11:25.701524Z","iopub.execute_input":"2022-06-22T09:11:25.701882Z","iopub.status.idle":"2022-06-22T09:11:34.772924Z","shell.execute_reply.started":"2022-06-22T09:11:25.701847Z","shell.execute_reply":"2022-06-22T09:11:34.771965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_y_pred = pd.DataFrame({\"target0_prob\":y_pred[:,0],\"target1_prob\":y_pred[:,1]})\ndf_y_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-22T09:11:34.774555Z","iopub.execute_input":"2022-06-22T09:11:34.775295Z","iopub.status.idle":"2022-06-22T09:11:34.797166Z","shell.execute_reply.started":"2022-06-22T09:11:34.775233Z","shell.execute_reply":"2022-06-22T09:11:34.796193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsub[\"prediction\"] = y_pred[:,1]\nsub.to_csv('submission.csv', index=False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T09:11:34.800303Z","iopub.execute_input":"2022-06-22T09:11:34.800811Z","iopub.status.idle":"2022-06-22T09:11:42.469844Z","shell.execute_reply.started":"2022-06-22T09:11:34.800759Z","shell.execute_reply":"2022-06-22T09:11:42.468653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}