{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# If it is useful, please vote","metadata":{}},{"cell_type":"markdown","source":"Add features referring to the following\nhttps://www.kaggle.com/code/manavtrivedi/tuffline-plotly-amex?scriptVersionId=102868130","metadata":{}},{"cell_type":"markdown","source":"# **Import**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n%matplotlib inline\nimport random\n\nimport warnings \nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:00.636853Z","iopub.execute_input":"2022-08-22T03:21:00.637308Z","iopub.status.idle":"2022-08-22T03:21:01.859440Z","shell.execute_reply.started":"2022-08-22T03:21:00.637222Z","shell.execute_reply":"2022-08-22T03:21:01.858522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_feather('../input/amexfeather/train_data.ftr')\ndf_train = df_train.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:01.861314Z","iopub.execute_input":"2022-08-22T03:21:01.861934Z","iopub.status.idle":"2022-08-22T03:21:29.591361Z","shell.execute_reply.started":"2022-08-22T03:21:01.861900Z","shell.execute_reply":"2022-08-22T03:21:29.589856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:29.594147Z","iopub.execute_input":"2022-08-22T03:21:29.595151Z","iopub.status.idle":"2022-08-22T03:21:29.608017Z","shell.execute_reply.started":"2022-08-22T03:21:29.595102Z","shell.execute_reply":"2022-08-22T03:21:29.606834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.dropna(axis=1, thresh=int(0.80 * len(df_train)))\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:29.611089Z","iopub.execute_input":"2022-08-22T03:21:29.611497Z","iopub.status.idle":"2022-08-22T03:21:30.643327Z","shell.execute_reply.started":"2022-08-22T03:21:29.611460Z","shell.execute_reply":"2022-08-22T03:21:30.642178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feature**","metadata":{}},{"cell_type":"code","source":"df_train[\"c_PD_239\"]=df_train[\"D_39\"]/(df_train[\"P_2\"]*(-1)+0.0001)\ndf_train[\"c_PB_29\"]=df_train[\"P_2\"]*(-1)/(df_train[\"B_9\"]*(1)+0.0001)\ndf_train[\"c_PR_21\"]=df_train[\"P_2\"]*(-1)/(df_train[\"R_1\"]+0.0001)\n\ndf_train[\"c_BBBB\"]=(df_train[\"B_9\"]+0.001)/(df_train[\"B_23\"]+df_train[\"B_3\"]+0.0001)\ndf_train[\"c_BBBB1\"]=(df_train[\"B_33\"]*(-1))+(df_train[\"B_18\"]*(-1)+df_train[\"S_25\"]*(1)+0.0001)\ndf_train[\"c_BBBB2\"]=(df_train[\"B_19\"]+df_train[\"B_20\"]+df_train[\"B_4\"]+0.0001)\n\ndf_train[\"c_RRR0\"]=(df_train[\"R_3\"]+0.001)/(df_train[\"R_2\"]+df_train[\"R_4\"]+0.0001)\ndf_train[\"c_RRR1\"]=(df_train[\"D_62\"]+0.001)/(df_train[\"D_112\"]+df_train[\"R_27\"]+0.0001)\n\ndf_train[\"c_PD_348\"]=df_train[\"D_48\"]/(df_train[\"P_3\"]+0.0001)\ndf_train[\"c_PD_355\"]=df_train[\"D_55\"]/(df_train[\"P_3\"]+0.0001)\n\ndf_train[\"c_PD_439\"]=df_train[\"D_39\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PB_49\"]=df_train[\"B_9\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PR_41\"]=df_train[\"R_1\"]/(df_train[\"P_4\"]+0.0001)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:30.645253Z","iopub.execute_input":"2022-08-22T03:21:30.645926Z","iopub.status.idle":"2022-08-22T03:21:30.976341Z","shell.execute_reply.started":"2022-08-22T03:21:30.645892Z","shell.execute_reply":"2022-08-22T03:21:30.975492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model&Predict**","metadata":{}},{"cell_type":"code","source":"y = df_train['target']\nX = df_train.drop(['target'],axis=1).drop(\"S_2\", axis=1)\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=26,stratify=y)\n\nprint(\"X_train Training Data Size :\",X_train.shape[0])\nprint(\"X_test Testing Data Size   :\",X_test.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:30.977758Z","iopub.execute_input":"2022-08-22T03:21:30.978990Z","iopub.status.idle":"2022-08-22T03:21:33.480299Z","shell.execute_reply.started":"2022-08-22T03:21:30.978951Z","shell.execute_reply":"2022-08-22T03:21:33.478857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xname = X.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:33.481897Z","iopub.execute_input":"2022-08-22T03:21:33.482985Z","iopub.status.idle":"2022-08-22T03:21:33.488117Z","shell.execute_reply.started":"2022-08-22T03:21:33.482940Z","shell.execute_reply":"2022-08-22T03:21:33.487131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna.integration.lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:33.489514Z","iopub.execute_input":"2022-08-22T03:21:33.490109Z","iopub.status.idle":"2022-08-22T03:21:35.412132Z","shell.execute_reply.started":"2022-08-22T03:21:33.490077Z","shell.execute_reply":"2022-08-22T03:21:35.411128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_SIZE = 0.4\nRANDOM_STATE = 42\n# trainのデータセットの3割をモデル学習時のバリデーションデータとして利用する\nX_train, X_valid, y_train, y_valid = train_test_split(X_train,\n                                                    y_train,\n                                                    test_size=TEST_SIZE,\n                                                    random_state=RANDOM_STATE)\n\n# LightGBMを利用するのに必要なフォーマットに変換\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_eval = lgb.Dataset(X_valid, y_valid, reference=lgb_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:35.413448Z","iopub.execute_input":"2022-08-22T03:21:35.413861Z","iopub.status.idle":"2022-08-22T03:21:36.560197Z","shell.execute_reply.started":"2022-08-22T03:21:35.413824Z","shell.execute_reply":"2022-08-22T03:21:36.558867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:36.564558Z","iopub.execute_input":"2022-08-22T03:21:36.564963Z","iopub.status.idle":"2022-08-22T03:21:36.570110Z","shell.execute_reply.started":"2022-08-22T03:21:36.564926Z","shell.execute_reply":"2022-08-22T03:21:36.568531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n# import optuna.integration.lightgbm as lgb\nmodel = lgb.LGBMClassifier(boosting_type='goss', min_depth=25, random_state=0,num_leaves=1000) #,num_leaves=200","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:22:01.094179Z","iopub.execute_input":"2022-08-22T03:22:01.094595Z","iopub.status.idle":"2022-08-22T03:22:01.100726Z","shell.execute_reply.started":"2022-08-22T03:22:01.094562Z","shell.execute_reply":"2022-08-22T03:22:01.099505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params = {\n#     'objective': 'mean_squared_error',\n#     'metric': 'mae',\n#     \"verbose\": -1,#verbosity\n#     \"verbose_eval\":-1,\n#     \"boosting_type\": \"gbdt\",\n# }\n\n# best_params, history = {}, []\n\n# # LightGBM学習\n# gbm = lgb.train(params,\n#                 lgb_train,\n#                 num_boost_round=200,\n#                 valid_sets=[lgb_train, lgb_eval],\n#                 early_stopping_rounds=50,\n#                 verbose_eval=False\n#                )\n\n# best_params = gbm.params\n# best_params","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:22:01.842684Z","iopub.execute_input":"2022-08-22T03:22:01.843109Z","iopub.status.idle":"2022-08-22T03:22:01.848909Z","shell.execute_reply.started":"2022-08-22T03:22:01.843075Z","shell.execute_reply":"2022-08-22T03:22:01.847675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = model.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:22:02.267748Z","iopub.execute_input":"2022-08-22T03:22:02.268549Z","iopub.status.idle":"2022-08-22T03:22:44.031139Z","shell.execute_reply.started":"2022-08-22T03:22:02.268497Z","shell.execute_reply":"2022-08-22T03:22:44.029850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:36.812086Z","iopub.status.idle":"2022-08-22T03:21:36.812540Z","shell.execute_reply.started":"2022-08-22T03:21:36.812320Z","shell.execute_reply":"2022-08-22T03:21:36.812338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submissions","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport glob\nfrom scipy.stats import rankdata\n\npaths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\ndfs = [pd.read_csv(x) for x in paths]\ndfs = [x.sort_values(by='customer_ID') for x in dfs]\n\npaths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\npaths\n\nfor df in dfs:\n    df['prediction'] = np.clip(df['prediction'], 0, 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:22:56.286688Z","iopub.execute_input":"2022-08-22T03:22:56.287131Z","iopub.status.idle":"2022-08-22T03:23:04.437223Z","shell.execute_reply.started":"2022-08-22T03:22:56.287094Z","shell.execute_reply":"2022-08-22T03:23:04.435967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\ndfs = [pd.read_csv(x) for x in paths]\ndfs = [x.sort_values(by='customer_ID') for x in dfs]","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:23:04.439186Z","iopub.execute_input":"2022-08-22T03:23:04.439546Z","iopub.status.idle":"2022-08-22T03:23:10.308899Z","shell.execute_reply.started":"2022-08-22T03:23:04.439516Z","shell.execute_reply":"2022-08-22T03:23:10.307830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = [0.22, 0.85, 0.97, 0.57, 1.02, 0.4]# [0.52, 0.87, 0.95, 0.57, 1, 0.8]","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:23:10.310389Z","iopub.execute_input":"2022-08-22T03:23:10.311027Z","iopub.status.idle":"2022-08-22T03:23:10.315656Z","shell.execute_reply.started":"2022-08-22T03:23:10.310981Z","shell.execute_reply":"2022-08-22T03:23:10.314817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfor df, weight in zip(dfs, weights):\n    submit['prediction'] += (df['prediction'] * weight)\n    \nsubmit['prediction'] /= np.sum(weights)\n\nsubmit.to_csv('mean_submission.csv', index=None)\n\n \nsubmit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfor df, weight in zip(dfs, weights):\n    submit['prediction'] += (rankdata(df['prediction'])/df.shape[0]) * weight\n    \nsubmit['prediction'] /= 4","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:23:10.317861Z","iopub.execute_input":"2022-08-22T03:23:10.318186Z","iopub.status.idle":"2022-08-22T03:23:17.168436Z","shell.execute_reply.started":"2022-08-22T03:23:10.318157Z","shell.execute_reply":"2022-08-22T03:23:17.167402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_feather('../input/amexfeather/test_data.ftr')\ndf_train = test.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:23:17.169598Z","iopub.execute_input":"2022-08-22T03:23:17.169915Z","iopub.status.idle":"2022-08-22T03:24:04.412951Z","shell.execute_reply.started":"2022-08-22T03:23:17.169886Z","shell.execute_reply":"2022-08-22T03:24:04.411832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:24:04.414516Z","iopub.execute_input":"2022-08-22T03:24:04.414977Z","iopub.status.idle":"2022-08-22T03:24:04.421572Z","shell.execute_reply.started":"2022-08-22T03:24:04.414934Z","shell.execute_reply":"2022-08-22T03:24:04.420827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.dropna(axis=1, thresh=int(0.80 * len(df_train)))\n\ndf_train[\"c_PD_239\"]=df_train[\"D_39\"]/(df_train[\"P_2\"]*(-1)+0.0001)\ndf_train[\"c_PB_29\"]=df_train[\"P_2\"]*(-1)/(df_train[\"B_9\"]*(1)+0.0001)\ndf_train[\"c_PR_21\"]=df_train[\"P_2\"]*(-1)/(df_train[\"R_1\"]+0.0001)\n\ndf_train[\"c_BBBB\"]=(df_train[\"B_9\"]+0.001)/(df_train[\"B_23\"]+df_train[\"B_3\"]+0.0001)\ndf_train[\"c_BBBB1\"]=(df_train[\"B_33\"]*(-1))+(df_train[\"B_18\"]*(-1)+df_train[\"S_25\"]*(1)+0.0001)\ndf_train[\"c_BBBB2\"]=(df_train[\"B_19\"]+df_train[\"B_20\"]+df_train[\"B_4\"]+0.0001)\n\ndf_train[\"c_RRR0\"]=(df_train[\"R_3\"]+0.001)/(df_train[\"R_2\"]+df_train[\"R_4\"]+0.0001)\ndf_train[\"c_RRR1\"]=(df_train[\"D_62\"]+0.001)/(df_train[\"D_112\"]+df_train[\"R_27\"]+0.0001)\n\ndf_train[\"c_PD_348\"]=df_train[\"D_48\"]/(df_train[\"P_3\"]+0.0001)\ndf_train[\"c_PD_355\"]=df_train[\"D_55\"]/(df_train[\"P_3\"]+0.0001)\n\ndf_train[\"c_PD_439\"]=df_train[\"D_39\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PB_49\"]=df_train[\"B_9\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PR_41\"]=df_train[\"R_1\"]/(df_train[\"P_4\"]+0.0001)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:24:04.423103Z","iopub.execute_input":"2022-08-22T03:24:04.423677Z","iopub.status.idle":"2022-08-22T03:24:07.072731Z","shell.execute_reply.started":"2022-08-22T03:24:04.423643Z","shell.execute_reply":"2022-08-22T03:24:07.071338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop(\"S_2\", axis=1)\nX = X[Xname]","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:24:07.076423Z","iopub.execute_input":"2022-08-22T03:24:07.076909Z","iopub.status.idle":"2022-08-22T03:24:08.402513Z","shell.execute_reply.started":"2022-08-22T03:24:07.076856Z","shell.execute_reply":"2022-08-22T03:24:08.401427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = model.predict_proba(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:24:08.404274Z","iopub.execute_input":"2022-08-22T03:24:08.404938Z","iopub.status.idle":"2022-08-22T03:24:13.482112Z","shell.execute_reply.started":"2022-08-22T03:24:08.404897Z","shell.execute_reply":"2022-08-22T03:24:13.480822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Y_pred = gbm.predict(X, num_iteration=gbm.best_iteration)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:24:13.486741Z","iopub.execute_input":"2022-08-22T03:24:13.487114Z","iopub.status.idle":"2022-08-22T03:24:13.491831Z","shell.execute_reply.started":"2022-08-22T03:24:13.487083Z","shell.execute_reply":"2022-08-22T03:24:13.490427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:24:13.493489Z","iopub.execute_input":"2022-08-22T03:24:13.494263Z","iopub.status.idle":"2022-08-22T03:24:13.509138Z","shell.execute_reply.started":"2022-08-22T03:24:13.494210Z","shell.execute_reply":"2022-08-22T03:24:13.507802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit['prediction'] = (submit['prediction'])*(0.99)+(Y_pred[:,0])*(0.01)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:36.836473Z","iopub.status.idle":"2022-08-22T03:21:36.836883Z","shell.execute_reply.started":"2022-08-22T03:21:36.836657Z","shell.execute_reply":"2022-08-22T03:21:36.836675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=None)    ","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:36.838752Z","iopub.status.idle":"2022-08-22T03:21:36.839174Z","shell.execute_reply.started":"2022-08-22T03:21:36.838978Z","shell.execute_reply":"2022-08-22T03:21:36.838997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-22T03:21:36.841031Z","iopub.status.idle":"2022-08-22T03:21:36.841970Z","shell.execute_reply.started":"2022-08-22T03:21:36.841694Z","shell.execute_reply":"2022-08-22T03:21:36.841722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}