{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport sys\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nimport gc\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import GridSearchCV\nimport math\nplt.style.use('ggplot')\nimport warnings as w\nw.filterwarnings(action='ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-15T15:41:01.040009Z","iopub.execute_input":"2022-06-15T15:41:01.040515Z","iopub.status.idle":"2022-06-15T15:41:01.974711Z","shell.execute_reply.started":"2022-06-15T15:41:01.040409Z","shell.execute_reply":"2022-06-15T15:41:01.973772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns',None)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:41:01.976757Z","iopub.execute_input":"2022-06-15T15:41:01.977448Z","iopub.status.idle":"2022-06-15T15:41:01.983918Z","shell.execute_reply.started":"2022-06-15T15:41:01.977407Z","shell.execute_reply":"2022-06-15T15:41:01.981909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_feather('../input/amexfeather/train_data.ftr')\ntrain = train.groupby('customer_ID').tail(1).set_index('customer_ID')\nprint(\"The training data begins on {} and ends on {}.\".format(train['S_2'].min().strftime('%m-%d-%Y'),train['S_2'].max().strftime('%m-%d-%Y')))\nprint(\"There are {:,.0f} customers in the training set and {} features.\".format(train.shape[0],train.shape[1]))\n\ntest = pd.read_feather('../input/amexfeather/test_data.ftr')\ntest = test.groupby('customer_ID').tail(1).set_index('customer_ID')\nprint(\"\\nThe test data begins on {} and ends on {}.\".format(test['S_2'].min().strftime('%m-%d-%Y'),test['S_2'].max().strftime('%m-%d-%Y')))\nprint(\"There are {:,.0f} customers in the test set and {} features.\".format(test.shape[0],test.shape[1]))\n\ndel test['S_2']\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:25:43.753222Z","iopub.execute_input":"2022-06-15T16:25:43.753578Z","iopub.status.idle":"2022-06-15T16:26:46.317476Z","shell.execute_reply.started":"2022-06-15T16:25:43.753549Z","shell.execute_reply":"2022-06-15T16:26:46.316528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### train set is date range 2018-03-01 ~ 2018-03-31 but test set is date range 2019-04-01 ~ 2019-10-31\n#### So it's difficult perfectly predict test customer credit default","metadata":{}},{"cell_type":"markdown","source":"### Feature Explain\n 1. D_* = Delinquency Variable (criminal?)\n 2. S_* = Spend Varibale \n 3. P_* = Payment Variable\n 4. B_* = Balance Variable\n 5. R_* = Risk variable\n \n### Categorical Variable\n   * 'B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'\n   \n#### Feature are all anonimized and normalized because of Personal information protection\n#### So I didn't known all original characteristics of features, So I know only the approximate characteristics Spend,Paymnet,Balance...etc","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"## Describe","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:06.979037Z","iopub.execute_input":"2022-06-15T15:42:06.979603Z","iopub.status.idle":"2022-06-15T15:42:20.307927Z","shell.execute_reply.started":"2022-06-15T15:42:06.979567Z","shell.execute_reply":"2022-06-15T15:42:20.307102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check Null Ratio","metadata":{}},{"cell_type":"code","source":"feature_null_ratio = round((train.isna().sum()/train.shape[0]*100),2).sort_values(ascending=False).astype(int)\nfeature_null_ratio = feature_null_ratio.to_frame().rename(columns={0:'Null Ratio(%)'})\nfeature_null_ratio.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:20.308916Z","iopub.execute_input":"2022-06-15T15:42:20.309254Z","iopub.status.idle":"2022-06-15T15:42:20.703300Z","shell.execute_reply.started":"2022-06-15T15:42:20.309219Z","shell.execute_reply":"2022-06-15T15:42:20.702441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In Top 20 missing value, Feature D is have the largest number of Null ratio","metadata":{}},{"cell_type":"markdown","source":"## Target count plot","metadata":{}},{"cell_type":"code","source":"ex = train.reset_index().groupby('S_2')['customer_ID'].nunique().reset_index()\nfig = go.Figure()\nfig.add_trace(\n    go.Scatter(x=ex.S_2,y=ex.customer_ID,hovertemplate='',mode='lines')\n)\nfig.update_layout(\n    title = 'Frequncy of customer statements',\n    xaxis_title = 'Date',\n    yaxis_title = 'satements update',\n    hovermode = 'x unified'\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:20.704605Z","iopub.execute_input":"2022-06-15T15:42:20.705916Z","iopub.status.idle":"2022-06-15T15:42:21.082204Z","shell.execute_reply.started":"2022-06-15T15:42:20.705872Z","shell.execute_reply":"2022-06-15T15:42:21.081462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### March 3,10,17,24 is all Saturday So We Knowing that In Saturday is highly increaseing customer statements ","metadata":{}},{"cell_type":"code","source":"del ex\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:21.083540Z","iopub.execute_input":"2022-06-15T15:42:21.084058Z","iopub.status.idle":"2022-06-15T15:42:21.181878Z","shell.execute_reply.started":"2022-06-15T15:42:21.084021Z","shell.execute_reply":"2022-06-15T15:42:21.181119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Target ValueCounts","metadata":{}},{"cell_type":"code","source":"train.target.value_counts(normalize=True).plot(kind='bar',figsize=(10,8),legend=True)\nprint(train.target.value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:21.184499Z","iopub.execute_input":"2022-06-15T15:42:21.185403Z","iopub.status.idle":"2022-06-15T15:42:21.412883Z","shell.execute_reply.started":"2022-06-15T15:42:21.185364Z","shell.execute_reply":"2022-06-15T15:42:21.411958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.target.value_counts(normalize=True).plot(kind='pie',figsize=(10,8),legend=True)\nprint(train.target.value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:21.414353Z","iopub.execute_input":"2022-06-15T15:42:21.414907Z","iopub.status.idle":"2022-06-15T15:42:21.590547Z","shell.execute_reply.started":"2022-06-15T15:42:21.414868Z","shell.execute_reply":"2022-06-15T15:42:21.589498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Target feature : Default customer is more than Normal customer","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:21.592268Z","iopub.execute_input":"2022-06-15T15:42:21.592703Z","iopub.status.idle":"2022-06-15T15:42:21.724720Z","shell.execute_reply.started":"2022-06-15T15:42:21.592662Z","shell.execute_reply":"2022-06-15T15:42:21.723897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Categorical Columns preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nencoder = LabelEncoder()\ncategorical_feature = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_68', 'D_64', 'D_66']\ncolumns = train.columns.values\nfor feature in categorical_feature:\n    if feature in columns:\n        train[feature] = encoder.fit_transform(train[feature])\n        test[feature] = encoder.fit_transform(test[feature])\n    else:\n        pass\n","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:26:46.323606Z","iopub.execute_input":"2022-06-15T16:26:46.325558Z","iopub.status.idle":"2022-06-15T16:26:47.745978Z","shell.execute_reply.started":"2022-06-15T16:26:46.325521Z","shell.execute_reply":"2022-06-15T16:26:47.745063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:26:47.747470Z","iopub.execute_input":"2022-06-15T16:26:47.747849Z","iopub.status.idle":"2022-06-15T16:26:48.150097Z","shell.execute_reply.started":"2022-06-15T16:26:47.747812Z","shell.execute_reply":"2022-06-15T16:26:48.149011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution","metadata":{}},{"cell_type":"markdown","source":"#### Delinquency","metadata":{}},{"cell_type":"code","source":"cols = [col for col in columns if (col.startswith(('D','T'))) & (col not in categorical_feature)]\ncols.append('target')\nex = train[cols]\nrow = 0\ntotal_row = math.ceil(len(cols) / 5)\ncol = [0,1,2,3,4] * total_row\nfig,ax = plt.subplots(total_row,5,figsize=(16,54))\nfig.suptitle('Distribution of Delinquency Variable',fontsize=16)\nfor i,feature in enumerate(ex.columns[:-1]):\n    if (i!=0) and (i%5==0):\n        row += 1\n    sns.kdeplot(x=feature,hue='target',data=ex,label=['Normal','Overdue'],fill=True,ax=ax[row,col[i]])\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(feature),ylabel=('Density' if i%5==0 else ''))\n    \nfor i in range(2,5):\n    ax[int(total_row)-1,i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.983))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])\n","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:42:23.318017Z","iopub.execute_input":"2022-06-15T15:42:23.318410Z","iopub.status.idle":"2022-06-15T15:44:58.376968Z","shell.execute_reply.started":"2022-06-15T15:42:23.318375Z","shell.execute_reply":"2022-06-15T15:44:58.375420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In D feature distribution distribution is nearly same default & normal but In Density D127,D123 default Density is bigger than Noraml Density\n#### So I think D127,D123 is more helpful to predict credit default and then other features","metadata":{}},{"cell_type":"code","source":"del ex\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:44:58.378247Z","iopub.execute_input":"2022-06-15T15:44:58.379050Z","iopub.status.idle":"2022-06-15T15:44:58.577031Z","shell.execute_reply.started":"2022-06-15T15:44:58.379013Z","shell.execute_reply":"2022-06-15T15:44:58.575963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spend","metadata":{}},{"cell_type":"code","source":"cols = [col for col in columns if (col.startswith(('S','T'))) & (col not in categorical_feature) & (col != 'S_2')]\ncols.append('target')\nex = train[cols]\nrow = 0\ntotal_row = math.ceil(len(cols) / 5)\ncol = [0,1,2,3,4] * total_row\nfig,ax = plt.subplots(total_row,5,figsize=(16,20))\nfig.suptitle('Distribution of Spend Variable',fontsize=16)\nfor i,feature in enumerate(ex.columns[:-1]):\n    if (i!=0) and (i%5==0):\n        row += 1\n    sns.kdeplot(x=feature,hue='target',data=ex,label=['Normal','Overdue'],fill=True,ax=ax[row,col[i]])\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(feature),ylabel=('Density' if i%5==0 else ''))\n    \nfor i in range(1,5):\n    ax[int(total_row-1),i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Normal'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.983))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])\n","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:44:58.578455Z","iopub.execute_input":"2022-06-15T15:44:58.578959Z","iopub.status.idle":"2022-06-15T15:45:42.563292Z","shell.execute_reply.started":"2022-06-15T15:44:58.578921Z","shell.execute_reply":"2022-06-15T15:45:42.562438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In Spend features S_16,S_26,S_24 Default Density is bigger than Normal Density ","metadata":{}},{"cell_type":"code","source":"del ex\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:45:42.564692Z","iopub.execute_input":"2022-06-15T15:45:42.565202Z","iopub.status.idle":"2022-06-15T15:45:42.891490Z","shell.execute_reply.started":"2022-06-15T15:45:42.565167Z","shell.execute_reply":"2022-06-15T15:45:42.890694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Risk","metadata":{}},{"cell_type":"code","source":"cols = [col for col in train.columns if (col.startswith(('R','T'))) & (col not in categorical_feature)]\ncols.append('target')\nex = train[cols]\nrow = 0\ntotal_row = math.ceil(len(cols) / 5)\nfig,ax = plt.subplots(total_row,5,figsize=(16,24))\nfig.suptitle('Distribution of Risk Variable',fontsize=16)\ncol = [0,1,2,3,4] * total_row\nfor i, feature in enumerate(ex.columns):\n    if (i!=0) & (i%5==0):\n        row+=1\n    sns.kdeplot(x=feature,hue='target',label=['Normal','Overdue'],fill=True,legend=False,\n                ax=ax[row,col[i]],data=ex)\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(feature),ylabel=('Density') if i%5==0 else '')\n    \nfor i in range(1,5):\n    ax[int(total_row-1),i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.984))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:45:42.892895Z","iopub.execute_input":"2022-06-15T15:45:42.893421Z","iopub.status.idle":"2022-06-15T15:46:41.239753Z","shell.execute_reply.started":"2022-06-15T15:45:42.893385Z","shell.execute_reply":"2022-06-15T15:46:41.238846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In Risk features R_20 is highly Density ","metadata":{}},{"cell_type":"code","source":"del ex\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:46:41.241182Z","iopub.execute_input":"2022-06-15T15:46:41.241681Z","iopub.status.idle":"2022-06-15T15:46:41.415567Z","shell.execute_reply.started":"2022-06-15T15:46:41.241647Z","shell.execute_reply":"2022-06-15T15:46:41.414796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Target Correlation","metadata":{}},{"cell_type":"code","source":"train.drop('S_2',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:27:41.228239Z","iopub.execute_input":"2022-06-15T16:27:41.228887Z","iopub.status.idle":"2022-06-15T16:27:41.629217Z","shell.execute_reply.started":"2022-06-15T16:27:41.228854Z","shell.execute_reply":"2022-06-15T16:27:41.628383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_feature = [cols for cols in train.columns if cols not in categorical_feature]\nfor feature in numeric_feature:\n    if train[feature][0].dtype == np.float16:\n        train[feature].fillna(-99.0,inplace=True)\n        test[feature].fillna(-99.0,inplace=True)\n    else:\n        pass\ntrain.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:27:41.739878Z","iopub.execute_input":"2022-06-15T16:27:41.740792Z","iopub.status.idle":"2022-06-15T16:27:43.272865Z","shell.execute_reply.started":"2022-06-15T16:27:41.740748Z","shell.execute_reply":"2022-06-15T16:27:43.271912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check top 20 positive & negative high correlation","metadata":{}},{"cell_type":"code","source":"corr_data = train[train.keys()]\ncmap = plt.cm.PuBu\ncols_positive = corr_data.corr().nlargest(20,'target')['target'].index\ncols_negative = corr_data.corr().nsmallest(20,'target')['target'].index\ncols = cols_positive.append(cols_negative)\ncm = np.corrcoef(corr_data[cols].values.T)\nfig,ax = plt.subplots(figsize=(25,20))\nsns.heatmap(cm,vmax=1,vmin=-1,square=True,annot=True,cmap=cmap,xticklabels=cols.values,yticklabels=cols.values)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:46:43.258458Z","iopub.execute_input":"2022-06-15T15:46:43.258913Z","iopub.status.idle":"2022-06-15T15:48:09.166116Z","shell.execute_reply.started":"2022-06-15T15:46:43.258874Z","shell.execute_reply":"2022-06-15T15:48:09.165450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### correlation is higher than 0.5(negative): B_18\n#### correlation is higher than 0.5(positive): B_9, B_23, D_75, D_58, B_7\n#### Generally correlation is higher than 50% is important(good) feature(columns)","metadata":{}},{"cell_type":"code","source":"del cm,corr_data,cols,cols_positive,cols_negative\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:48:09.167219Z","iopub.execute_input":"2022-06-15T15:48:09.168033Z","iopub.status.idle":"2022-06-15T15:48:09.345613Z","shell.execute_reply.started":"2022-06-15T15:48:09.167996Z","shell.execute_reply":"2022-06-15T15:48:09.344861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model & train & valid split\n\n#### I thought train way SMOTE(oversampling) & Normal \n#### The reason why I applied SMOTE because of taget feature data is very imbalance ","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:48:09.346726Z","iopub.execute_input":"2022-06-15T15:48:09.347328Z","iopub.status.idle":"2022-06-15T15:48:09.504060Z","shell.execute_reply.started":"2022-06-15T15:48:09.347289Z","shell.execute_reply":"2022-06-15T15:48:09.503241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Before apply oversampling(SMOTE) train & valid data split\n### Because I don't want affect validation data cause using SMOTE ","metadata":{}},{"cell_type":"code","source":"train_idx = int(len(train) * 0.8)\nvalid_idx = len(train) - train_idx\nprint(train_idx,valid_idx)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:40.346562Z","iopub.execute_input":"2022-06-15T16:11:40.347003Z","iopub.status.idle":"2022-06-15T16:11:40.352900Z","shell.execute_reply.started":"2022-06-15T16:11:40.346965Z","shell.execute_reply":"2022-06-15T16:11:40.351981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = train[:train_idx]\nvalid_set = train[-valid_idx:]\nprint(train_set.shape,valid_set.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:40.354567Z","iopub.execute_input":"2022-06-15T16:11:40.355188Z","iopub.status.idle":"2022-06-15T16:11:40.362769Z","shell.execute_reply.started":"2022-06-15T16:11:40.355151Z","shell.execute_reply":"2022-06-15T16:11:40.361843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:40.364657Z","iopub.execute_input":"2022-06-15T16:11:40.365385Z","iopub.status.idle":"2022-06-15T16:11:40.581589Z","shell.execute_reply.started":"2022-06-15T16:11:40.365330Z","shell.execute_reply":"2022-06-15T16:11:40.580687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_set.pop('target')\nx_train = train_set\nprint(x_train.shape,y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:40.583260Z","iopub.execute_input":"2022-06-15T16:11:40.583801Z","iopub.status.idle":"2022-06-15T16:11:40.592698Z","shell.execute_reply.started":"2022-06-15T16:11:40.583765Z","shell.execute_reply":"2022-06-15T16:11:40.591840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val = valid_set.pop('target')\nx_val = valid_set","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:40.593929Z","iopub.execute_input":"2022-06-15T16:11:40.595099Z","iopub.status.idle":"2022-06-15T16:11:40.600159Z","shell.execute_reply.started":"2022-06-15T16:11:40.595065Z","shell.execute_reply":"2022-06-15T16:11:40.599214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_set,valid_set\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:40.602688Z","iopub.execute_input":"2022-06-15T16:11:40.603108Z","iopub.status.idle":"2022-06-15T16:11:40.812366Z","shell.execute_reply.started":"2022-06-15T16:11:40.603072Z","shell.execute_reply":"2022-06-15T16:11:40.811313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nsmote = SMOTE()\n\nx_over_train, y_over_train = smote.fit_resample(x_train.values,y_train.values)\nprint(x_over_train.shape, y_over_train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:48:09.913581Z","iopub.execute_input":"2022-06-15T15:48:09.913989Z","iopub.status.idle":"2022-06-15T15:51:50.979136Z","shell.execute_reply.started":"2022-06-15T15:48:09.913950Z","shell.execute_reply":"2022-06-15T15:51:50.978318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(y_over_train).value_counts(normalize=True).plot(kind='bar')\nprint(pd.DataFrame(y_over_train).value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:50.980322Z","iopub.execute_input":"2022-06-15T15:51:50.980892Z","iopub.status.idle":"2022-06-15T15:51:51.184546Z","shell.execute_reply.started":"2022-06-15T15:51:50.980855Z","shell.execute_reply":"2022-06-15T15:51:51.183798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier,Dataset,early_stopping,log_evaluation\nfrom lightgbm import plot_importance,plot_metric\nfrom sklearn.metrics import accuracy_score,roc_auc_score,r2_score","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:51.188372Z","iopub.execute_input":"2022-06-15T15:51:51.190364Z","iopub.status.idle":"2022-06-15T15:51:53.292491Z","shell.execute_reply.started":"2022-06-15T15:51:51.190325Z","shell.execute_reply":"2022-06-15T15:51:53.291547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def metrics(y_true: pd.DataFrame, pred: pd.DataFrame) -> float:\n    \n    def top_foure_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = pd.concat([y_true,pred],axis='columns').sort_values('prediction',ascending=False)\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \n    def weighted_gini(y_true:pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true,pred],axis='columns').sort_values('prediction',ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n    \n    def normalized_weighted_gini(y_true: pd.DataFrame, pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target':'prediction'})\n        return weighted_gini(y_true,pred) / weighted_gini(y_true,pred)\n    \n    G = normalized_weighted_gini(y_true,pred)\n    D = top_foure_percent_captured(y_true,pred)\n    \n    return 0.5 * (G+D)\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.294168Z","iopub.execute_input":"2022-06-15T15:51:53.294593Z","iopub.status.idle":"2022-06-15T15:51:53.308637Z","shell.execute_reply.started":"2022-06-15T15:51:53.294556Z","shell.execute_reply":"2022-06-15T15:51:53.307637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Grid Search","metadata":{}},{"cell_type":"code","source":"# n_estimators = [1000,2000]\n# max_depth = [1,100,200,300]\n# learning_rate = [0.03,0.05,0.08]\n# reg_alpha = [0.001,0.01,0.1]\n# reg_lambda = [0.001,0.01,0.1]\n# subsample = [0.88]\n","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.310316Z","iopub.execute_input":"2022-06-15T15:51:53.310723Z","iopub.status.idle":"2022-06-15T15:51:53.319591Z","shell.execute_reply.started":"2022-06-15T15:51:53.310687Z","shell.execute_reply":"2022-06-15T15:51:53.318746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params =  {\n#     'n_estimators' : n_estimators,\n#     'max_depth': max_depth,\n#     'learning_rate' : learning_rate,\n#     'reg_alpha' : reg_alpha,\n#     'reg_lambda' : reg_lambda,\n#     'subsample' : subsample,\n#     'n_jobs' : [-1]\n# }","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.321440Z","iopub.execute_input":"2022-06-15T15:51:53.322058Z","iopub.status.idle":"2022-06-15T15:51:53.328433Z","shell.execute_reply.started":"2022-06-15T15:51:53.322021Z","shell.execute_reply":"2022-06-15T15:51:53.327455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# gsc = GridSearchCV(LGBMClassifier(device='gpu',objective='binary',boosting_type='gbdt')\n#                    ,param_grid=params,verbose=10,return_train_score=True,\n#                    scoring='roc_auc',cv=3,n_jobs=-1)\n# gsc.fit(x_over_train,y_over_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.330023Z","iopub.execute_input":"2022-06-15T15:51:53.330564Z","iopub.status.idle":"2022-06-15T15:51:53.336986Z","shell.execute_reply.started":"2022-06-15T15:51:53.330524Z","shell.execute_reply":"2022-06-15T15:51:53.335801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(gsc.best_params_)\n# params = gsc.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.338358Z","iopub.execute_input":"2022-06-15T15:51:53.339142Z","iopub.status.idle":"2022-06-15T15:51:53.346016Z","shell.execute_reply.started":"2022-06-15T15:51:53.339099Z","shell.execute_reply":"2022-06-15T15:51:53.345124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scoreing(fold,y_true,y_pred):\n    acc = accuracy_score(y_true,y_pred)\n    auc_score = roc_auc_score(y_true,y_pred)\n    R2 = r2_score(y_true,y_pred)\n    y_true = pd.DataFrame(data={'target':y_true.reset_index(drop=True)})\n    y_pred = pd.DataFrame(data={'prediction':y_pred})\n    gini_score = metrics(y_true,y_pred)\n    print('Fold{}\\tAccuracy:{:.3f}\\tR2:{:.3f}\\tAUC:{:.3f}\\tGini:{:.3f}'.format(fold,acc,R2,auc_score,gini_score))\n    return gini_score\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.349707Z","iopub.execute_input":"2022-06-15T15:51:53.350062Z","iopub.status.idle":"2022-06-15T15:51:53.356893Z","shell.execute_reply.started":"2022-06-15T15:51:53.350029Z","shell.execute_reply":"2022-06-15T15:51:53.355770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kfold training","metadata":{}},{"cell_type":"code","source":" params = {\n     'boosting_type': 'gbdt',\n     'n_estimators': 5000,\n     'num_leaves': 50,\n     'learning_rate': 0.05,\n     'colsample_bytree': 0.9,\n     'min_child_samples': 2000,\n     'reg_alpha': 2,\n     'objective': 'binary',\n     'random_state': 21,\n     'device': 'gpu',\n     'n_jobs': -1,\n     'subsample': 0.88,\n     'max_depth': 100\n          }","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.358295Z","iopub.execute_input":"2022-06-15T15:51:53.359021Z","iopub.status.idle":"2022-06-15T15:51:53.366386Z","shell.execute_reply.started":"2022-06-15T15:51:53.358928Z","shell.execute_reply":"2022-06-15T15:51:53.365265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SMOTE dataset(train)","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=5,shuffle=True)\nfold = 1\nlgb_over_models = []\nfor train_idx,valid_idx in skf.split(x_over_train,y_over_train):\n    print('-'*58)\n    print(f'Fold:{fold}')\n    train_x, valid_x = x_over_train[train_idx], x_over_train[valid_idx]\n    train_y, valid_y = y_over_train[train_idx], y_over_train[valid_idx]\n    model = LGBMClassifier(**params)\n    model.fit(train_x,train_y,eval_set=[(valid_x,valid_y)],\n              callbacks=[early_stopping(200)],\n              verbose=200,eval_metric=['binary_logloss','auc'])\n    pred = model.predict_proba(x_val)\n    lgb_over_models.append(model)\n    gini_score = scoreing(fold,y_val,pred)\n    if fold == 1:\n        best_over_gini_score = gini_score\n        best_over_fold = fold\n    else:\n        if gini_score > best_over_gini_score:\n            best_over_fold = fold\n            best_over_gini_score = gini_score\n    plot_metric(model)\n    fold += 1\nprint(f'Best Fold:{best_over_fold}\\tBest Gini score:{best_over_gini_score}')\nprint(f'So We using {best_over_fold} model')    ","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:51:53.367942Z","iopub.execute_input":"2022-06-15T15:51:53.368734Z","iopub.status.idle":"2022-06-15T16:04:56.014371Z","shell.execute_reply.started":"2022-06-15T15:51:53.368665Z","shell.execute_reply":"2022-06-15T16:04:56.013563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x_over_train,y_over_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:04:56.015796Z","iopub.execute_input":"2022-06-15T16:04:56.016182Z","iopub.status.idle":"2022-06-15T16:04:56.265252Z","shell.execute_reply.started":"2022-06-15T16:04:56.016146Z","shell.execute_reply":"2022-06-15T16:04:56.264192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normal dataset","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=5,shuffle=True)\nlgb_models = []\nfor fold ,(train_idx,valid_idx) in enumerate(skf.split(x_train,y_train)):\n    print('-'*58)\n    print(f'Fold:{fold+1}')\n    train_x, valid_x = x_train.values[train_idx], x_train.values[valid_idx]\n    train_y, valid_y = y_train[train_idx], y_train[valid_idx]\n    model = LGBMClassifier(**params)\n    model.fit(train_x,train_y,eval_set=[(valid_x,valid_y)],\n              callbacks=[early_stopping(200)],\n              verbose=200,eval_metric=['binary_logloss','auc'])\n    pred = model.predict_proba(x_val)\n    lgb_models.append(model)\n    gini_score = scoreing(fold,y_val,pred)\n    if fold == 0:\n        best_gini_score = gini_score\n        best_fold = fold\n    else:\n        if gini_score > best_gini_score:\n            best_fold = fold\n            best_gini_score = gini_score\n    plot_metric(model)\nprint(f'Best Fold:{best_fold}\\tBest Gini score:{best_gini_score}')\nprint(f'So We using {best_fold} model')    ","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:11:49.826514Z","iopub.execute_input":"2022-06-15T16:11:49.827334Z","iopub.status.idle":"2022-06-15T16:17:31.638892Z","shell.execute_reply.started":"2022-06-15T16:11:49.827293Z","shell.execute_reply":"2022-06-15T16:17:31.637947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"cell_type":"code","source":"def feature_importance(model,train):\n    plt.figure(figsize=(15,10))\n    feature = pd.Series(model.feature_importances_,index=train.columns)\n    sort_feature = feature.sort_values(ascending=False)[:30]\n    return sns.barplot(x=sort_feature,y=sort_feature.index)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:17:34.971280Z","iopub.execute_input":"2022-06-15T16:17:34.971788Z","iopub.status.idle":"2022-06-15T16:17:34.978525Z","shell.execute_reply.started":"2022-06-15T16:17:34.971755Z","shell.execute_reply":"2022-06-15T16:17:34.977503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_over_fold","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:17:49.755962Z","iopub.execute_input":"2022-06-15T16:17:49.756439Z","iopub.status.idle":"2022-06-15T16:17:49.762720Z","shell.execute_reply.started":"2022-06-15T16:17:49.756407Z","shell.execute_reply":"2022-06-15T16:17:49.761788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Oversampling result\nbest_over_model = lgb_over_models[best_over_fold-1]\nfeature_importance(best_over_model,x_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:17:55.598823Z","iopub.execute_input":"2022-06-15T16:17:55.599180Z","iopub.status.idle":"2022-06-15T16:17:56.077892Z","shell.execute_reply.started":"2022-06-15T16:17:55.599151Z","shell.execute_reply":"2022-06-15T16:17:56.077108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = lgb_models[best_fold]\nfeature_importance(best_model,x_train)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:18:09.112713Z","iopub.execute_input":"2022-06-15T16:18:09.113144Z","iopub.status.idle":"2022-06-15T16:18:09.489353Z","shell.execute_reply.started":"2022-06-15T16:18:09.113108Z","shell.execute_reply":"2022-06-15T16:18:09.488524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_over = best_model.predict_proba(test.values)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:27:55.860497Z","iopub.execute_input":"2022-06-15T16:27:55.861093Z","iopub.status.idle":"2022-06-15T16:28:41.917296Z","shell.execute_reply.started":"2022-06-15T16:27:55.861059Z","shell.execute_reply":"2022-06-15T16:28:41.916614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = best_model.predict_proba(test.values)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:28:41.919011Z","iopub.execute_input":"2022-06-15T16:28:41.919607Z","iopub.status.idle":"2022-06-15T16:29:29.130077Z","shell.execute_reply.started":"2022-06-15T16:28:41.919571Z","shell.execute_reply":"2022-06-15T16:29:29.128779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:20:03.994700Z","iopub.execute_input":"2022-06-15T16:20:03.995046Z","iopub.status.idle":"2022-06-15T16:20:04.273795Z","shell.execute_reply.started":"2022-06-15T16:20:03.995012Z","shell.execute_reply":"2022-06-15T16:20:04.272877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:20:04.275804Z","iopub.execute_input":"2022-06-15T16:20:04.276433Z","iopub.status.idle":"2022-06-15T16:20:05.867304Z","shell.execute_reply.started":"2022-06-15T16:20:04.276391Z","shell.execute_reply":"2022-06-15T16:20:05.866510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['prediction'] = pred\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:30:23.674183Z","iopub.execute_input":"2022-06-15T16:30:23.674548Z","iopub.status.idle":"2022-06-15T16:30:23.695047Z","shell.execute_reply.started":"2022-06-15T16:30:23.674515Z","shell.execute_reply":"2022-06-15T16:30:23.694288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:30:36.023779Z","iopub.execute_input":"2022-06-15T16:30:36.024488Z","iopub.status.idle":"2022-06-15T16:30:40.747494Z","shell.execute_reply.started":"2022-06-15T16:30:36.024455Z","shell.execute_reply":"2022-06-15T16:30:40.746528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['prediction'] = pred_over\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:32:49.218508Z","iopub.execute_input":"2022-06-15T16:32:49.219072Z","iopub.status.idle":"2022-06-15T16:32:49.240435Z","shell.execute_reply.started":"2022-06-15T16:32:49.219034Z","shell.execute_reply":"2022-06-15T16:32:49.239519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:24:23.136211Z","iopub.execute_input":"2022-06-15T16:24:23.136779Z","iopub.status.idle":"2022-06-15T16:24:26.418219Z","shell.execute_reply.started":"2022-06-15T16:24:23.136733Z","shell.execute_reply":"2022-06-15T16:24:26.417214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-15T16:05:55.153158Z","iopub.status.idle":"2022-06-15T16:05:55.154182Z","shell.execute_reply.started":"2022-06-15T16:05:55.153903Z","shell.execute_reply":"2022-06-15T16:05:55.153929Z"},"trusted":true},"execution_count":null,"outputs":[]}]}