{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3727003,"sourceType":"datasetVersion","datasetId":2213609},{"sourceId":100354158,"sourceType":"kernelVersion"},{"sourceId":101187485,"sourceType":"kernelVersion"},{"sourceId":101322601,"sourceType":"kernelVersion"}],"dockerImageVersionId":30213,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# If it is useful, please vote","metadata":{}},{"cell_type":"markdown","source":"Add features referring to the following\nhttps://www.kaggle.com/code/manavtrivedi/tuffline-plotly-amex?scriptVersionId=102868130","metadata":{}},{"cell_type":"markdown","source":"# **Import**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n%matplotlib inline\nimport random\n\nimport warnings \nwarnings.filterwarnings('ignore')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_feather('../input/amexfeather/train_data.ftr')\ndf_train = df_train.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:28.634991Z","iopub.execute_input":"2022-08-13T04:08:28.635822Z","iopub.status.idle":"2022-08-13T04:08:49.340948Z","shell.execute_reply.started":"2022-08-13T04:08:28.635773Z","shell.execute_reply":"2022-08-13T04:08:49.335347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:49.346388Z","iopub.execute_input":"2022-08-13T04:08:49.34785Z","iopub.status.idle":"2022-08-13T04:08:49.358636Z","shell.execute_reply.started":"2022-08-13T04:08:49.347792Z","shell.execute_reply":"2022-08-13T04:08:49.35717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.dropna(axis=1, thresh=int(0.80 * len(df_train)))\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:49.360127Z","iopub.execute_input":"2022-08-13T04:08:49.361079Z","iopub.status.idle":"2022-08-13T04:08:51.333606Z","shell.execute_reply.started":"2022-08-13T04:08:49.361016Z","shell.execute_reply":"2022-08-13T04:08:51.332072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feature**","metadata":{}},{"cell_type":"code","source":"df_train[\"c_PD_239\"]=df_train[\"D_39\"]/(df_train[\"P_2\"]*(-1)+0.0001)\ndf_train[\"c_PB_29\"]=df_train[\"P_2\"]*(-1)/(df_train[\"B_9\"]*(1)+0.0001)\ndf_train[\"c_PR_21\"]=df_train[\"P_2\"]*(-1)/(df_train[\"R_1\"]+0.0001)\n\ndf_train[\"c_BBBB\"]=(df_train[\"B_9\"]+0.001)/(df_train[\"B_23\"]+df_train[\"B_3\"]+0.0001)\ndf_train[\"c_BBBB1\"]=(df_train[\"B_33\"]*(-1))+(df_train[\"B_18\"]*(-1)+df_train[\"S_25\"]*(1)+0.0001)\ndf_train[\"c_BBBB2\"]=(df_train[\"B_19\"]+df_train[\"B_20\"]+df_train[\"B_4\"]+0.0001)\n\ndf_train[\"c_RRR0\"]=(df_train[\"R_3\"]+0.001)/(df_train[\"R_2\"]+df_train[\"R_4\"]+0.0001)\ndf_train[\"c_RRR1\"]=(df_train[\"D_62\"]+0.001)/(df_train[\"D_112\"]+df_train[\"R_27\"]+0.0001)\n\ndf_train[\"c_PD_348\"]=df_train[\"D_48\"]/(df_train[\"P_3\"]+0.0001)\ndf_train[\"c_PD_355\"]=df_train[\"D_55\"]/(df_train[\"P_3\"]+0.0001)\n\ndf_train[\"c_PD_439\"]=df_train[\"D_39\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PB_49\"]=df_train[\"B_9\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PR_41\"]=df_train[\"R_1\"]/(df_train[\"P_4\"]+0.0001)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:51.336894Z","iopub.execute_input":"2022-08-13T04:08:51.337315Z","iopub.status.idle":"2022-08-13T04:08:51.70169Z","shell.execute_reply.started":"2022-08-13T04:08:51.337279Z","shell.execute_reply":"2022-08-13T04:08:51.700243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model&Predict**","metadata":{}},{"cell_type":"code","source":"y = df_train['target']\nX = df_train.drop(['target'],axis=1).drop(\"S_2\", axis=1)\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=26,stratify=y)\n\nprint(\"X_train Training Data Size :\",X_train.shape[0])\nprint(\"X_test Testing Data Size   :\",X_test.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:51.816374Z","iopub.execute_input":"2022-08-13T04:08:51.817663Z","iopub.status.idle":"2022-08-13T04:08:54.113471Z","shell.execute_reply.started":"2022-08-13T04:08:51.817613Z","shell.execute_reply":"2022-08-13T04:08:54.112134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xname = X.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:54.114933Z","iopub.execute_input":"2022-08-13T04:08:54.115328Z","iopub.status.idle":"2022-08-13T04:08:54.12084Z","shell.execute_reply.started":"2022-08-13T04:08:54.115293Z","shell.execute_reply":"2022-08-13T04:08:54.119592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:54.122482Z","iopub.execute_input":"2022-08-13T04:08:54.123548Z","iopub.status.idle":"2022-08-13T04:08:54.135107Z","shell.execute_reply.started":"2022-08-13T04:08:54.1235Z","shell.execute_reply":"2022-08-13T04:08:54.134186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nmodel = lgb.LGBMClassifier(boosting_type='goss', max_depth=5, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:54.137658Z","iopub.execute_input":"2022-08-13T04:08:54.138195Z","iopub.status.idle":"2022-08-13T04:08:54.154075Z","shell.execute_reply.started":"2022-08-13T04:08:54.138123Z","shell.execute_reply":"2022-08-13T04:08:54.152984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:08:54.156358Z","iopub.execute_input":"2022-08-13T04:08:54.157229Z","iopub.status.idle":"2022-08-13T04:09:11.132195Z","shell.execute_reply.started":"2022-08-13T04:08:54.157187Z","shell.execute_reply":"2022-08-13T04:09:11.130927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submissions","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport glob\nfrom scipy.stats import rankdata\n\npaths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\ndfs = [pd.read_csv(x) for x in paths]\ndfs = [x.sort_values(by='customer_ID') for x in dfs]\n\npaths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\npaths\n\nfor df in dfs:\n    df['prediction'] = np.clip(df['prediction'], 0, 1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = [x for x in glob.glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\ndfs = [pd.read_csv(x) for x in paths]\ndfs = [x.sort_values(by='customer_ID') for x in dfs]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = [0.52, 0.85, 0.97, 0.57, 1.02, 0.8]# [0.52, 0.87, 0.95, 0.57, 1, 0.8]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:09:39.634722Z","iopub.execute_input":"2022-08-13T04:09:39.635501Z","iopub.status.idle":"2022-08-13T04:09:39.644079Z","shell.execute_reply.started":"2022-08-13T04:09:39.635453Z","shell.execute_reply":"2022-08-13T04:09:39.642465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfor df, weight in zip(dfs, weights):\n    submit['prediction'] += (df['prediction'] * weight)\n    \nsubmit['prediction'] /= np.sum(weights)\n\nsubmit.to_csv('mean_submission.csv', index=None)\n\n \nsubmit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmit['prediction'] = 0\n\nfor df, weight in zip(dfs, weights):\n    submit['prediction'] += (rankdata(df['prediction'])/df.shape[0]) * weight\n    \nsubmit['prediction'] /= 4","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:09:39.646073Z","iopub.execute_input":"2022-08-13T04:09:39.647215Z","iopub.status.idle":"2022-08-13T04:09:45.855311Z","shell.execute_reply.started":"2022-08-13T04:09:39.647162Z","shell.execute_reply":"2022-08-13T04:09:45.854025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_feather('../input/amexfeather/test_data.ftr')\ndf_train = test.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:09:45.865483Z","iopub.execute_input":"2022-08-13T04:09:45.866346Z","iopub.status.idle":"2022-08-13T04:10:34.454959Z","shell.execute_reply.started":"2022-08-13T04:09:45.866279Z","shell.execute_reply":"2022-08-13T04:10:34.453704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:10:34.460106Z","iopub.execute_input":"2022-08-13T04:10:34.460503Z","iopub.status.idle":"2022-08-13T04:10:34.469858Z","shell.execute_reply.started":"2022-08-13T04:10:34.460468Z","shell.execute_reply":"2022-08-13T04:10:34.468339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.dropna(axis=1, thresh=int(0.80 * len(df_train)))\n\ndf_train[\"c_PD_239\"]=df_train[\"D_39\"]/(df_train[\"P_2\"]*(-1)+0.0001)\ndf_train[\"c_PB_29\"]=df_train[\"P_2\"]*(-1)/(df_train[\"B_9\"]*(1)+0.0001)\ndf_train[\"c_PR_21\"]=df_train[\"P_2\"]*(-1)/(df_train[\"R_1\"]+0.0001)\n\ndf_train[\"c_BBBB\"]=(df_train[\"B_9\"]+0.001)/(df_train[\"B_23\"]+df_train[\"B_3\"]+0.0001)\ndf_train[\"c_BBBB1\"]=(df_train[\"B_33\"]*(-1))+(df_train[\"B_18\"]*(-1)+df_train[\"S_25\"]*(1)+0.0001)\ndf_train[\"c_BBBB2\"]=(df_train[\"B_19\"]+df_train[\"B_20\"]+df_train[\"B_4\"]+0.0001)\n\ndf_train[\"c_RRR0\"]=(df_train[\"R_3\"]+0.001)/(df_train[\"R_2\"]+df_train[\"R_4\"]+0.0001)\ndf_train[\"c_RRR1\"]=(df_train[\"D_62\"]+0.001)/(df_train[\"D_112\"]+df_train[\"R_27\"]+0.0001)\n\ndf_train[\"c_PD_348\"]=df_train[\"D_48\"]/(df_train[\"P_3\"]+0.0001)\ndf_train[\"c_PD_355\"]=df_train[\"D_55\"]/(df_train[\"P_3\"]+0.0001)\n\ndf_train[\"c_PD_439\"]=df_train[\"D_39\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PB_49\"]=df_train[\"B_9\"]/(df_train[\"P_4\"]+0.0001)\ndf_train[\"c_PR_41\"]=df_train[\"R_1\"]/(df_train[\"P_4\"]+0.0001)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:10:34.471784Z","iopub.execute_input":"2022-08-13T04:10:34.472622Z","iopub.status.idle":"2022-08-13T04:10:36.787885Z","shell.execute_reply.started":"2022-08-13T04:10:34.47257Z","shell.execute_reply":"2022-08-13T04:10:36.786906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop(\"S_2\", axis=1)\nX = X[Xname]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:10:36.789245Z","iopub.execute_input":"2022-08-13T04:10:36.789596Z","iopub.status.idle":"2022-08-13T04:10:38.164362Z","shell.execute_reply.started":"2022-08-13T04:10:36.789564Z","shell.execute_reply":"2022-08-13T04:10:38.163168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = model.predict_proba(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:10:38.165836Z","iopub.execute_input":"2022-08-13T04:10:38.166324Z","iopub.status.idle":"2022-08-13T04:10:43.185283Z","shell.execute_reply.started":"2022-08-13T04:10:38.166271Z","shell.execute_reply":"2022-08-13T04:10:43.184285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit['prediction'] = (submit['prediction'])*(0.99)+(Y_pred[:,0])*(0.01)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:11:51.375139Z","iopub.execute_input":"2022-08-13T04:11:51.377735Z","iopub.status.idle":"2022-08-13T04:11:51.457066Z","shell.execute_reply.started":"2022-08-13T04:11:51.377044Z","shell.execute_reply":"2022-08-13T04:11:51.445125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=None)    ","metadata":{"execution":{"iopub.status.busy":"2022-08-13T04:10:43.599762Z","iopub.status.idle":"2022-08-13T04:10:43.600695Z","shell.execute_reply.started":"2022-08-13T04:10:43.600452Z","shell.execute_reply":"2022-08-13T04:10:43.600477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}