{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-30T12:44:09.400865Z","iopub.execute_input":"2022-06-30T12:44:09.401245Z","iopub.status.idle":"2022-06-30T12:44:09.413442Z","shell.execute_reply.started":"2022-06-30T12:44:09.401213Z","shell.execute_reply":"2022-06-30T12:44:09.412491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_feather('../input/amex-default-prediction-feather/train.feather')\nlabels_df = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\nsub_df = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:07:04.324339Z","iopub.execute_input":"2022-06-30T13:07:04.324769Z","iopub.status.idle":"2022-06-30T13:07:23.662238Z","shell.execute_reply.started":"2022-06-30T13:07:04.324734Z","shell.execute_reply":"2022-06-30T13:07:23.660572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport itertools\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import NullFormatter\nimport pandas as pd\nimport numpy as np\nimport matplotlib.ticker as ticker\nfrom sklearn import preprocessing\n\n\n# visualization tools\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedKFold \nfrom sklearn.metrics import roc_auc_score, roc_curve, auc","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:32.468670Z","iopub.execute_input":"2022-06-30T12:44:32.469164Z","iopub.status.idle":"2022-06-30T12:44:33.398409Z","shell.execute_reply.started":"2022-06-30T12:44:32.469120Z","shell.execute_reply":"2022-06-30T12:44:33.397190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:33.402380Z","iopub.execute_input":"2022-06-30T12:44:33.402894Z","iopub.status.idle":"2022-06-30T12:44:33.430800Z","shell.execute_reply.started":"2022-06-30T12:44:33.402847Z","shell.execute_reply":"2022-06-30T12:44:33.429512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:33.435284Z","iopub.execute_input":"2022-06-30T12:44:33.436331Z","iopub.status.idle":"2022-06-30T12:44:33.458185Z","shell.execute_reply.started":"2022-06-30T12:44:33.436283Z","shell.execute_reply":"2022-06-30T12:44:33.456580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:33.462378Z","iopub.execute_input":"2022-06-30T12:44:33.462855Z","iopub.status.idle":"2022-06-30T12:44:37.417844Z","shell.execute_reply.started":"2022-06-30T12:44:33.462813Z","shell.execute_reply":"2022-06-30T12:44:37.415923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****DATA PREPARATION****","metadata":{}},{"cell_type":"code","source":"list(train_df[:1000].columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:37.420240Z","iopub.execute_input":"2022-06-30T12:44:37.420631Z","iopub.status.idle":"2022-06-30T12:44:37.433587Z","shell.execute_reply.started":"2022-06-30T12:44:37.420599Z","shell.execute_reply":"2022-06-30T12:44:37.431891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df.customer_ID.unique())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:37.435338Z","iopub.execute_input":"2022-06-30T12:44:37.435777Z","iopub.status.idle":"2022-06-30T12:44:38.303737Z","shell.execute_reply.started":"2022-06-30T12:44:37.435730Z","shell.execute_reply":"2022-06-30T12:44:38.302668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.customer_ID.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:38.305174Z","iopub.execute_input":"2022-06-30T12:44:38.305490Z","iopub.status.idle":"2022-06-30T12:44:38.977409Z","shell.execute_reply.started":"2022-06-30T12:44:38.305461Z","shell.execute_reply":"2022-06-30T12:44:38.976204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('customer_ID').size()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:38.979237Z","iopub.execute_input":"2022-06-30T12:44:38.980028Z","iopub.status.idle":"2022-06-30T12:44:40.348878Z","shell.execute_reply.started":"2022-06-30T12:44:38.979989Z","shell.execute_reply":"2022-06-30T12:44:40.347670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:40.350825Z","iopub.execute_input":"2022-06-30T12:44:40.351566Z","iopub.status.idle":"2022-06-30T12:44:46.055710Z","shell.execute_reply.started":"2022-06-30T12:44:40.351528Z","shell.execute_reply":"2022-06-30T12:44:46.054357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perc = 75.0\nmin_count =  int(((100-perc)/100)*train_df.shape[0] + 1)\ndf = train_df.dropna( axis=1, \n                thresh=min_count)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:07:23.664389Z","iopub.execute_input":"2022-06-30T13:07:23.664771Z","iopub.status.idle":"2022-06-30T13:07:42.886282Z","shell.execute_reply.started":"2022-06-30T13:07:23.664734Z","shell.execute_reply":"2022-06-30T13:07:42.884560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:44:56.977344Z","iopub.execute_input":"2022-06-30T12:44:56.977624Z","iopub.status.idle":"2022-06-30T12:44:58.993738Z","shell.execute_reply.started":"2022-06-30T12:44:56.977598Z","shell.execute_reply":"2022-06-30T12:44:58.992539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = df.columns\ncols = [cols[0]]+sorted(cols[1:-1],key = lambda k:k[0])+[cols[-1]]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:07:42.890569Z","iopub.execute_input":"2022-06-30T13:07:42.891168Z","iopub.status.idle":"2022-06-30T13:07:42.899923Z","shell.execute_reply.started":"2022-06-30T13:07:42.891111Z","shell.execute_reply":"2022-06-30T13:07:42.898059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[cols]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:07:42.904809Z","iopub.execute_input":"2022-06-30T13:07:42.905587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:03.235534Z","iopub.execute_input":"2022-06-30T12:45:03.236006Z","iopub.status.idle":"2022-06-30T12:45:04.918710Z","shell.execute_reply.started":"2022-06-30T12:45:03.235962Z","shell.execute_reply":"2022-06-30T12:45:04.917564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.mode.chained_assignment = None\ndf['S_2'] = pd.to_datetime(df['S_2'])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:04.920193Z","iopub.execute_input":"2022-06-30T12:45:04.920655Z","iopub.status.idle":"2022-06-30T12:45:05.656326Z","shell.execute_reply.started":"2022-06-30T12:45:04.920621Z","shell.execute_reply":"2022-06-30T12:45:05.655212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.mode.chained_assignment = None\ndf['dayofweek'] = df['S_2'].dt.dayofweek\ndf['month'] = df['S_2'].dt.month\ndf['year'] = df['S_2'].dt.year","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:05.657760Z","iopub.execute_input":"2022-06-30T12:45:05.658068Z","iopub.status.idle":"2022-06-30T12:45:07.269175Z","shell.execute_reply.started":"2022-06-30T12:45:05.658038Z","shell.execute_reply":"2022-06-30T12:45:07.268137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= df.drop('S_2',axis =1)\ndf['weekend'] = df['dayofweek'].apply(lambda x: 1 if (x>3)  else 0)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:07.270425Z","iopub.execute_input":"2022-06-30T12:45:07.270743Z","iopub.status.idle":"2022-06-30T12:45:14.146000Z","shell.execute_reply.started":"2022-06-30T12:45:07.270715Z","shell.execute_reply":"2022-06-30T12:45:14.144941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df[:200000]\ndf1.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:14.147464Z","iopub.execute_input":"2022-06-30T12:45:14.147802Z","iopub.status.idle":"2022-06-30T12:45:20.185127Z","shell.execute_reply.started":"2022-06-30T12:45:14.147773Z","shell.execute_reply":"2022-06-30T12:45:20.183786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:20.187137Z","iopub.execute_input":"2022-06-30T12:45:20.187617Z","iopub.status.idle":"2022-06-30T12:45:20.219452Z","shell.execute_reply.started":"2022-06-30T12:45:20.187568Z","shell.execute_reply":"2022-06-30T12:45:20.218209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df = labels_df.set_index('customer_ID')\n","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:20.221911Z","iopub.execute_input":"2022-06-30T12:45:20.222385Z","iopub.status.idle":"2022-06-30T12:45:20.239832Z","shell.execute_reply.started":"2022-06-30T12:45:20.222338Z","shell.execute_reply":"2022-06-30T12:45:20.238563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:20.241671Z","iopub.execute_input":"2022-06-30T12:45:20.242389Z","iopub.status.idle":"2022-06-30T12:45:20.255276Z","shell.execute_reply.started":"2022-06-30T12:45:20.242342Z","shell.execute_reply":"2022-06-30T12:45:20.254114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Total Number of Unique Customers: ',len(df['customer_ID'].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:20.256981Z","iopub.execute_input":"2022-06-30T12:45:20.258006Z","iopub.status.idle":"2022-06-30T12:45:21.072815Z","shell.execute_reply.started":"2022-06-30T12:45:20.257959Z","shell.execute_reply":"2022-06-30T12:45:21.071723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.mode.chained_assignment = None\nfill_target = lambda col: labels_df.loc[col,'target']\ndf1['target'] = df1['customer_ID'].apply(fill_target)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:21.074279Z","iopub.execute_input":"2022-06-30T12:45:21.074703Z","iopub.status.idle":"2022-06-30T12:45:22.893960Z","shell.execute_reply.started":"2022-06-30T12:45:21.074656Z","shell.execute_reply":"2022-06-30T12:45:22.892711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:22.898357Z","iopub.execute_input":"2022-06-30T12:45:22.898753Z","iopub.status.idle":"2022-06-30T12:45:22.907233Z","shell.execute_reply.started":"2022-06-30T12:45:22.898716Z","shell.execute_reply":"2022-06-30T12:45:22.905772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:22.908815Z","iopub.execute_input":"2022-06-30T12:45:22.909893Z","iopub.status.idle":"2022-06-30T12:45:23.069741Z","shell.execute_reply.started":"2022-06-30T12:45:22.909852Z","shell.execute_reply":"2022-06-30T12:45:23.068883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Categorical = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68','target']","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:23.070942Z","iopub.execute_input":"2022-06-30T12:45:23.071444Z","iopub.status.idle":"2022-06-30T12:45:23.076503Z","shell.execute_reply.started":"2022-06-30T12:45:23.071414Z","shell.execute_reply":"2022-06-30T12:45:23.075150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[Categorical]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:23.078241Z","iopub.execute_input":"2022-06-30T12:45:23.079117Z","iopub.status.idle":"2022-06-30T12:45:23.109933Z","shell.execute_reply.started":"2022-06-30T12:45:23.079063Z","shell.execute_reply":"2022-06-30T12:45:23.108746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.groupby('customer_ID').tail(1).set_index('customer_ID')\ndf = df.merge(labels_df, on='customer_ID', how='left')","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:23.111444Z","iopub.execute_input":"2022-06-30T12:45:23.111817Z","iopub.status.idle":"2022-06-30T12:45:26.672237Z","shell.execute_reply.started":"2022-06-30T12:45:23.111780Z","shell.execute_reply":"2022-06-30T12:45:26.671137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nimputer=SimpleImputer(strategy=\"most_frequent\")\nImputed = pd.DataFrame(imputer.fit_transform(df[Categorical]),columns = Categorical)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:26.673684Z","iopub.execute_input":"2022-06-30T12:45:26.674448Z","iopub.status.idle":"2022-06-30T12:45:27.180487Z","shell.execute_reply.started":"2022-06-30T12:45:26.674402Z","shell.execute_reply":"2022-06-30T12:45:27.179089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[Categorical] = Imputed\ndf[Categorical] = df[Categorical].fillna(df[Categorical].mode())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:27.181889Z","iopub.execute_input":"2022-06-30T12:45:27.182206Z","iopub.status.idle":"2022-06-30T12:45:27.463974Z","shell.execute_reply.started":"2022-06-30T12:45:27.182178Z","shell.execute_reply":"2022-06-30T12:45:27.462863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[Categorical].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:27.465267Z","iopub.execute_input":"2022-06-30T12:45:27.465610Z","iopub.status.idle":"2022-06-30T12:45:27.487288Z","shell.execute_reply.started":"2022-06-30T12:45:27.465570Z","shell.execute_reply":"2022-06-30T12:45:27.486068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Numerical = df.select_dtypes(np.number).columns","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:27.489123Z","iopub.execute_input":"2022-06-30T12:45:27.489582Z","iopub.status.idle":"2022-06-30T12:45:27.840335Z","shell.execute_reply.started":"2022-06-30T12:45:27.489537Z","shell.execute_reply":"2022-06-30T12:45:27.839211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Numerical)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:27.842107Z","iopub.execute_input":"2022-06-30T12:45:27.842595Z","iopub.status.idle":"2022-06-30T12:45:27.848576Z","shell.execute_reply.started":"2022-06-30T12:45:27.842546Z","shell.execute_reply":"2022-06-30T12:45:27.847463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****EXPLORATORY DATA ANALYSIS****","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\ndf1[Categorical].hist(figsize = (20,20))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:27.849988Z","iopub.execute_input":"2022-06-30T12:45:27.850334Z","iopub.status.idle":"2022-06-30T12:45:29.807250Z","shell.execute_reply.started":"2022-06-30T12:45:27.850304Z","shell.execute_reply":"2022-06-30T12:45:29.806092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn\nseaborn.heatmap(df1[Categorical])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T12:45:33.780533Z","iopub.execute_input":"2022-06-30T12:45:33.780981Z","iopub.status.idle":"2022-06-30T12:45:36.207882Z","shell.execute_reply.started":"2022-06-30T12:45:33.780945Z","shell.execute_reply":"2022-06-30T12:45:36.206659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"background_color = 'white'\nmissing = pd.DataFrame(columns = ['% Missing values'],data = df.isnull().sum()/len(df_train))\nfig = plt.figure(figsize = (20, 60),facecolor=background_color)\ngs = fig.add_gridspec(1, 2)\ngs.update(wspace = 0.5, hspace = 0.5)\nax0 = fig.add_subplot(gs[0, 0])\nfor s in [\"right\", \"top\",\"bottom\",\"left\"]:\n    ax0.spines[s].set_visible(False)\nsns.heatmap(missing,cbar = False,annot = True,fmt =\".2%\", linewidths = 2,cmap = custom_colors,vmax = 1, ax = ax0)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\ndf1['target'].hist(figsize = (5,5))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:17.108915Z","iopub.execute_input":"2022-06-30T13:02:17.109558Z","iopub.status.idle":"2022-06-30T13:02:17.293718Z","shell.execute_reply.started":"2022-06-30T13:02:17.109520Z","shell.execute_reply":"2022-06-30T13:02:17.292788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df1.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:17.514611Z","iopub.execute_input":"2022-06-30T13:02:17.515862Z","iopub.status.idle":"2022-06-30T13:02:20.003171Z","shell.execute_reply.started":"2022-06-30T13:02:17.515807Z","shell.execute_reply":"2022-06-30T13:02:20.002016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Delinquency = [ cols for cols in df1.columns if cols[0]=='D']\nSpend = [ cols for cols in df1.columns if cols[0]=='S']\nPayment = [ cols for cols in df1.columns if cols[0]=='P']\nBalance = [ cols for cols in df1.columns if cols[0]=='B']\nRisk = [ cols for cols in df1.columns if cols[0]=='R']","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:20.004988Z","iopub.execute_input":"2022-06-30T13:02:20.005329Z","iopub.status.idle":"2022-06-30T13:02:20.012876Z","shell.execute_reply.started":"2022-06-30T13:02:20.005299Z","shell.execute_reply":"2022-06-30T13:02:20.011631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=['Delinquency', 'Spend','Payment','Balance','Risk']\nvalues= [len(Delinquency), len(Spend),len(Payment), len(Balance),len(Risk)]\nfig_1 = go.Figure()\nfig_1.add_trace(go.Pie(values = values,labels = labels,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig_1.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#70d6ff\",\"#ff9770\"]),\n                  title = dict(text = 'Feature Distribution'))  \nfig_1.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:05:29.839068Z","iopub.execute_input":"2022-06-30T13:05:29.839529Z","iopub.status.idle":"2022-06-30T13:05:29.887824Z","shell.execute_reply.started":"2022-06-30T13:05:29.839484Z","shell.execute_reply":"2022-06-30T13:05:29.886858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[Spend].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:20.014631Z","iopub.execute_input":"2022-06-30T13:02:20.015348Z","iopub.status.idle":"2022-06-30T13:02:20.099495Z","shell.execute_reply.started":"2022-06-30T13:02:20.015299Z","shell.execute_reply":"2022-06-30T13:02:20.098503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[df1[Spend].sum(axis=1)>=df1[Balance].sum(axis=1)]","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:20.101397Z","iopub.execute_input":"2022-06-30T13:02:20.101980Z","iopub.status.idle":"2022-06-30T13:02:20.329304Z","shell.execute_reply.started":"2022-06-30T13:02:20.101935Z","shell.execute_reply":"2022-06-30T13:02:20.327785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in Delinquency:\n    print(df1[col].unique())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:20.331400Z","iopub.execute_input":"2022-06-30T13:02:20.331755Z","iopub.status.idle":"2022-06-30T13:02:20.621360Z","shell.execute_reply.started":"2022-06-30T13:02:20.331717Z","shell.execute_reply":"2022-06-30T13:02:20.620073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(Delinquency)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:20.676626Z","iopub.execute_input":"2022-06-30T13:02:20.677060Z","iopub.status.idle":"2022-06-30T13:02:20.684985Z","shell.execute_reply.started":"2022-06-30T13:02:20.677023Z","shell.execute_reply":"2022-06-30T13:02:20.683832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\ndf1[df1[Delinquency].sum(axis=1)>=79]['target'].hist(figsize = (5,5))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:21.279031Z","iopub.execute_input":"2022-06-30T13:02:21.279425Z","iopub.status.idle":"2022-06-30T13:02:21.722048Z","shell.execute_reply.started":"2022-06-30T13:02:21.279393Z","shell.execute_reply":"2022-06-30T13:02:21.720706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perc_1= df1.loc[df1[Delinquency].sum(axis=1)>=79]['target'].sum()/df1.loc[df1[Delinquency].sum(axis=1)>=79].shape[0]\nprint('Average Percentage of Defaults for people who have more than nominal Deliquency measure: ', perc_1*100)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:21.848579Z","iopub.execute_input":"2022-06-30T13:02:21.849373Z","iopub.status.idle":"2022-06-30T13:02:22.412626Z","shell.execute_reply.started":"2022-06-30T13:02:21.849331Z","shell.execute_reply":"2022-06-30T13:02:22.411232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\ndf1[df1[Spend].sum(axis=1)>=df1[Balance].sum(axis=1)]['target'].hist(figsize = (5,5))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:22.414510Z","iopub.execute_input":"2022-06-30T13:02:22.414971Z","iopub.status.idle":"2022-06-30T13:02:22.821478Z","shell.execute_reply.started":"2022-06-30T13:02:22.414937Z","shell.execute_reply":"2022-06-30T13:02:22.820511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perc_2 = df1[df1[Spend].sum(axis=1)>=df1[Balance].sum(axis=1)]['target'].sum()/df1[df1[Spend].sum(axis=1)>=df1[Balance].sum(axis=1)].shape[0]\nprint('Average Percentage of Defaults for people who have more spend value than balance: ', perc_2*100)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:22.901272Z","iopub.execute_input":"2022-06-30T13:02:22.902014Z","iopub.status.idle":"2022-06-30T13:02:23.355459Z","shell.execute_reply.started":"2022-06-30T13:02:22.901962Z","shell.execute_reply":"2022-06-30T13:02:23.353045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in Risk:\n    print(df1[col].unique())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:23.426954Z","iopub.execute_input":"2022-06-30T13:02:23.428487Z","iopub.status.idle":"2022-06-30T13:02:23.553909Z","shell.execute_reply.started":"2022-06-30T13:02:23.428411Z","shell.execute_reply":"2022-06-30T13:02:23.552716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(Risk)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:23.742658Z","iopub.execute_input":"2022-06-30T13:02:23.743982Z","iopub.status.idle":"2022-06-30T13:02:23.751297Z","shell.execute_reply.started":"2022-06-30T13:02:23.743934Z","shell.execute_reply":"2022-06-30T13:02:23.750415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\ndf1.loc[df1[Risk].sum(axis=1)>=26]['target'].hist(figsize = (5,5))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:24.510676Z","iopub.execute_input":"2022-06-30T13:02:24.512075Z","iopub.status.idle":"2022-06-30T13:02:24.797120Z","shell.execute_reply.started":"2022-06-30T13:02:24.512028Z","shell.execute_reply":"2022-06-30T13:02:24.795839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perc_3= df1.loc[df1[Risk].sum(axis=1)>=26]['target'].sum()/df1.loc[df1[Risk].sum(axis=1)>=26].shape[0]\nprint('Average Percentage of Defaults for people who have more than nominal Risk measure: ', perc_3*100)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:24.813021Z","iopub.execute_input":"2022-06-30T13:02:24.813402Z","iopub.status.idle":"2022-06-30T13:02:24.990443Z","shell.execute_reply.started":"2022-06-30T13:02:24.813371Z","shell.execute_reply":"2022-06-30T13:02:24.989143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in Payment:\n    print(df1[col].unique())","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:25.567607Z","iopub.execute_input":"2022-06-30T13:02:25.568378Z","iopub.status.idle":"2022-06-30T13:02:25.588574Z","shell.execute_reply.started":"2022-06-30T13:02:25.568314Z","shell.execute_reply":"2022-06-30T13:02:25.587448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,15))\ndf1.loc[df1[Payment].sum(axis=1)>=1.5]['target'].hist(figsize = (5,5))","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:26.509853Z","iopub.execute_input":"2022-06-30T13:02:26.510246Z","iopub.status.idle":"2022-06-30T13:02:26.839032Z","shell.execute_reply.started":"2022-06-30T13:02:26.510213Z","shell.execute_reply":"2022-06-30T13:02:26.837590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perc_4= df1.loc[df1[Payment].sum(axis=1)>=1.5]['target'].sum()/df1.loc[df1[Payment].sum(axis=1)>=1.5].shape[0]\nprint('Average Percentage of Defaults for people who have more than nominal Payment: ', perc_4*100)","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:27.068107Z","iopub.execute_input":"2022-06-30T13:02:27.068546Z","iopub.status.idle":"2022-06-30T13:02:27.284619Z","shell.execute_reply.started":"2022-06-30T13:02:27.068503Z","shell.execute_reply":"2022-06-30T13:02:27.282666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****FEATURE SELECTION****","metadata":{}},{"cell_type":"code","source":"import plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\ncorr = df.corrwith(df['target'], axis=0)\nval = [str(round(v ,2) *100) + '%' for v in corr.values]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(y=corr.index, x= corr.values,\n                     orientation='h',\n                     marker_color = '#9900cc',\n                     text = val,\n                     textposition = 'outside',\n                     textfont_color = '#ffff80'))\nfig.update_layout(template = 'plotly_dark',\n                  title = \"Correlation with Target\",\n                  width = 800,\n                  height = 3000)\nfig.update_xaxes(range=[-2,2])","metadata":{"execution":{"iopub.status.busy":"2022-06-30T13:02:28.196786Z","iopub.execute_input":"2022-06-30T13:02:28.197239Z","iopub.status.idle":"2022-06-30T13:02:30.084307Z","shell.execute_reply.started":"2022-06-30T13:02:28.197198Z","shell.execute_reply":"2022-06-30T13:02:30.083115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [corr.index[i] for i in range(corr.shape[0]) if abs(corr[i])>0.095 and corr.index[i]!='target']\n#features = [col for col in df.columns if col not in ['target','weekend','year','month','dayofweek']]\n# all_cols = df.columns\n# non_use_cols = ['S_2','B_30','B_38','D_114','D_116','D_117','D_120','D_126','D_63','D_64','D_66','D_68', 'target','customer_ID','weekend','year','month','dayofweek']\n# features = [col for col in all_cols if col not in non_use_cols]\n","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:07.810338Z","iopub.execute_input":"2022-06-28T16:39:07.811313Z","iopub.status.idle":"2022-06-28T16:39:07.820988Z","shell.execute_reply.started":"2022-06-28T16:39:07.811271Z","shell.execute_reply":"2022-06-28T16:39:07.819419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:07.82384Z","iopub.execute_input":"2022-06-28T16:39:07.824209Z","iopub.status.idle":"2022-06-28T16:39:07.83428Z","shell.execute_reply.started":"2022-06-28T16:39:07.824181Z","shell.execute_reply":"2022-06-28T16:39:07.833002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outlier_list = []\noutlier_col = []\n\nfor col in features:\n    \n    temp_df = df[(df[col] > df[col].mean() + df[col].std() * 200) |\n                       (df[col] < df[col].mean() - df[col].std() * 200) ]\n    temp_df.head()\n    if len(temp_df) >0 and len(temp_df) <6 : \n        outliers = temp_df.index.to_list()\n        outlier_list.extend(outliers)\n        outlier_col.append(col)\n        print(col, len(temp_df))\n    \noutlier_list = list(set(outlier_list))\n","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:07.835454Z","iopub.execute_input":"2022-06-28T16:39:07.836333Z","iopub.status.idle":"2022-06-28T16:39:41.215087Z","shell.execute_reply.started":"2022-06-28T16:39:07.836298Z","shell.execute_reply":"2022-06-28T16:39:41.21396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(outlier_list)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:41.216565Z","iopub.execute_input":"2022-06-28T16:39:41.217077Z","iopub.status.idle":"2022-06-28T16:39:41.222046Z","shell.execute_reply.started":"2022-06-28T16:39:41.21705Z","shell.execute_reply":"2022-06-28T16:39:41.221082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**MODEL**","metadata":{}},{"cell_type":"code","source":"def amex_metric_official(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:41.223837Z","iopub.execute_input":"2022-06-28T16:39:41.22438Z","iopub.status.idle":"2022-06-28T16:39:41.237941Z","shell.execute_reply.started":"2022-06-28T16:39:41.224338Z","shell.execute_reply":"2022-06-28T16:39:41.236796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df[features]\ny=df['target']","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:41.239889Z","iopub.execute_input":"2022-06-28T16:39:41.240552Z","iopub.status.idle":"2022-06-28T16:39:41.399222Z","shell.execute_reply.started":"2022-06-28T16:39:41.240519Z","shell.execute_reply":"2022-06-28T16:39:41.397609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.apply(pd.to_numeric, errors='coerce')\nfor col in features:\n    X[col]  = X[col].fillna(np.mean([float(i) for i in X[col].dropna()]))","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:41.401197Z","iopub.execute_input":"2022-06-28T16:39:41.402077Z","iopub.status.idle":"2022-06-28T16:39:41.406555Z","shell.execute_reply.started":"2022-06-28T16:39:41.402036Z","shell.execute_reply":"2022-06-28T16:39:41.405815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.pipeline import Pipeline, FeatureUnion\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:41.407601Z","iopub.execute_input":"2022-06-28T16:39:41.408682Z","iopub.status.idle":"2022-06-28T16:39:41.50196Z","shell.execute_reply.started":"2022-06-28T16:39:41.408641Z","shell.execute_reply":"2022-06-28T16:39:41.501081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.33,random_state=100)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:39:41.503108Z","iopub.execute_input":"2022-06-28T16:39:41.503343Z","iopub.status.idle":"2022-06-28T16:39:42.997425Z","shell.execute_reply.started":"2022-06-28T16:39:41.503321Z","shell.execute_reply":"2022-06-28T16:39:42.995117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score\nimport lightgbm as lgbm\n\nkf = KFold(n_splits = 5)\nmodels = []\nlgbm_params ={\"objective\":\"binary\",\n              \"random_seed\":1234,\n             \"n_estimators\" : 2000}\n\nfor train_index, val_index in kf.split(X):\n    X_train = X.iloc[train_index]\n    X_valid = X.iloc[val_index]\n    Y_train = y.iloc[train_index]\n    Y_valid = y.iloc[val_index]\n    \n    lgbm_train = lgbm.Dataset(X_train, Y_train)\n    lgbm_eval = lgbm.Dataset(X_valid, Y_valid, reference=lgbm_train)\n    \n    model_lgbm = lgbm.train(lgbm_params,\n                           lgbm_train,\n                           valid_sets = lgbm_eval,\n                           num_boost_round = 300,\n                           early_stopping_rounds = 20,\n                           verbose_eval = 10,\n                           )\n    y_pred = model_lgbm.predict(X_valid, num_iteration = model_lgbm.best_iteration)\n    \n    \n    models.append(model_lgbm)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:46:48.630153Z","iopub.execute_input":"2022-06-28T16:46:48.63051Z","iopub.status.idle":"2022-06-28T16:49:33.84196Z","shell.execute_reply.started":"2022-06-28T16:46:48.630483Z","shell.execute_reply":"2022-06-28T16:49:33.841191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Extra Trees Classiifer**","metadata":{}},{"cell_type":"code","source":"clf = ExtraTreesClassifier()\n\n\nclf = GridSearchCV(\n    estimator=clf,\n    param_grid={\n        'n_estimators': [10],\n        'max_features': [50],\n        #'min_samples_leaf': range(20,50,5),\n        #'min_samples_split': range(15,36,5),\n    },\n    scoring='r2',\n    cv=5\n","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.687289Z","iopub.status.idle":"2022-06-28T16:42:27.688452Z","shell.execute_reply.started":"2022-06-28T16:42:27.688278Z","shell.execute_reply":"2022-06-28T16:42:27.688298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****SMOTE RANDOMFOREST****","metadata":{}},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nfrom imblearn.pipeline import Pipeline\n\nrf_clf = RandomForestClassifier(criterion='gini', bootstrap=True, random_state=100,n_estimators = 500)\nsmote_sampler = SMOTE(random_state=9)\nclf = Pipeline(steps = [['smote', smote_sampler],\n                             ['classifier', rf_clf]])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.689546Z","iopub.status.idle":"2022-06-28T16:42:27.689948Z","shell.execute_reply.started":"2022-06-28T16:42:27.689737Z","shell.execute_reply":"2022-06-28T16:42:27.689778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclf.fit(X_train,y_train)\ny_pred1 = clf.predict(X_test)\ny_pred_prob1 = clf.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.692031Z","iopub.status.idle":"2022-06-28T16:42:27.692331Z","shell.execute_reply.started":"2022-06-28T16:42:27.692189Z","shell.execute_reply":"2022-06-28T16:42:27.692204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns=[\"target\"])\ny_pred1 = pd.DataFrame(y_pred1, columns=[\"prediction\"])\ny_pred_prob1 = pd.DataFrame(y_pred_prob1, columns=[\"prediction\"])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.693175Z","iopub.status.idle":"2022-06-28T16:42:27.693437Z","shell.execute_reply.started":"2022-06-28T16:42:27.693307Z","shell.execute_reply":"2022-06-28T16:42:27.69332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric_official(y_test, y_pred_prob1))","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.695358Z","iopub.status.idle":"2022-06-28T16:42:27.695639Z","shell.execute_reply.started":"2022-06-28T16:42:27.695507Z","shell.execute_reply":"2022-06-28T16:42:27.69552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\naccuracy = metrics.accuracy_score(y_test[\"target\"], y_pred1[\"prediction\"])\naccuracy","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.696608Z","iopub.status.idle":"2022-06-28T16:42:27.69691Z","shell.execute_reply.started":"2022-06-28T16:42:27.696734Z","shell.execute_reply":"2022-06-28T16:42:27.696789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Bagging Classifier****","metadata":{}},{"cell_type":"code","source":"bag = BaggingClassifier(KNeighborsClassifier(),\n                             max_samples=0.5, max_features=0.5)\nbag.fit(X_train,y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.697489Z","iopub.status.idle":"2022-06-28T16:42:27.69775Z","shell.execute_reply.started":"2022-06-28T16:42:27.697613Z","shell.execute_reply":"2022-06-28T16:42:27.697626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = bag.predict(X_test) # predict for the test set\ny_pred_prob = bag.predict_proba(X_test)[:,1]\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.699248Z","iopub.status.idle":"2022-06-28T16:42:27.699518Z","shell.execute_reply.started":"2022-06-28T16:42:27.699383Z","shell.execute_reply":"2022-06-28T16:42:27.699396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns=[\"target\"])\ny_pred = pd.DataFrame(y_pred, columns=[\"prediction\"])\ny_pred_prob = pd.DataFrame(y_pred_prob, columns=[\"prediction\"])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.701033Z","iopub.status.idle":"2022-06-28T16:42:27.701499Z","shell.execute_reply.started":"2022-06-28T16:42:27.70129Z","shell.execute_reply":"2022-06-28T16:42:27.701311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.703393Z","iopub.status.idle":"2022-06-28T16:42:27.703805Z","shell.execute_reply.started":"2022-06-28T16:42:27.703579Z","shell.execute_reply":"2022-06-28T16:42:27.703599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_prob","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.704873Z","iopub.status.idle":"2022-06-28T16:42:27.705249Z","shell.execute_reply.started":"2022-06-28T16:42:27.705063Z","shell.execute_reply":"2022-06-28T16:42:27.705082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric_official(y_test, y_pred_prob))","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.706851Z","iopub.status.idle":"2022-06-28T16:42:27.707667Z","shell.execute_reply.started":"2022-06-28T16:42:27.707188Z","shell.execute_reply":"2022-06-28T16:42:27.707235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\naccuracy = metrics.accuracy_score(y_test[\"target\"], y_pred[\"prediction\"])\naccuracy","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.709088Z","iopub.status.idle":"2022-06-28T16:42:27.709528Z","shell.execute_reply.started":"2022-06-28T16:42:27.709322Z","shell.execute_reply":"2022-06-28T16:42:27.709343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****XGBOOST****","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nxgb_model = xgb.XGBClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.711347Z","iopub.status.idle":"2022-06-28T16:42:27.711814Z","shell.execute_reply.started":"2022-06-28T16:42:27.711582Z","shell.execute_reply":"2022-06-28T16:42:27.711604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameters = {'nthread':[4], #when use hyperthread, xgboost may become slower\n              'objective':['binary:logistic'],\n              'learning_rate': [0.05], #so called `eta` value\n              'max_depth': [6],\n              'min_child_weight': [11],\n              'silent': [1],\n              'subsample': [0.8],\n              'colsample_bytree': [0.7],\n              'n_estimators': [200], #number of trees, change it to 1000 for better results\n              'missing':[-999],\n              'seed': [1337]}","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.713077Z","iopub.status.idle":"2022-06-28T16:42:27.713497Z","shell.execute_reply.started":"2022-06-28T16:42:27.713286Z","shell.execute_reply":"2022-06-28T16:42:27.713308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = GridSearchCV(xgb_model, parameters, n_jobs=5, \n                   cv=StratifiedKFold(n_splits=2, shuffle=True), \n                   scoring='roc_auc',\n                   verbose=2, refit=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.714863Z","iopub.status.idle":"2022-06-28T16:42:27.715263Z","shell.execute_reply.started":"2022-06-28T16:42:27.71506Z","shell.execute_reply":"2022-06-28T16:42:27.71508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.fit(X_train,y_train)\ny_pred1 = clf.predict(X_test)\ny_pred_prob1 = clf.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.71624Z","iopub.status.idle":"2022-06-28T16:42:27.716605Z","shell.execute_reply.started":"2022-06-28T16:42:27.716415Z","shell.execute_reply":"2022-06-28T16:42:27.716434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns=[\"target\"])\ny_pred1 = pd.DataFrame(y_pred1, columns=[\"prediction\"])\ny_pred_prob1 = pd.DataFrame(y_pred_prob1, columns=[\"prediction\"])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.718728Z","iopub.status.idle":"2022-06-28T16:42:27.719081Z","shell.execute_reply.started":"2022-06-28T16:42:27.718938Z","shell.execute_reply":"2022-06-28T16:42:27.718954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric_official(y_test, y_pred_prob1))","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.720122Z","iopub.status.idle":"2022-06-28T16:42:27.720406Z","shell.execute_reply.started":"2022-06-28T16:42:27.720265Z","shell.execute_reply":"2022-06-28T16:42:27.72028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\naccuracy = metrics.accuracy_score(y_test[\"target\"], y_pred1[\"prediction\"])\naccuracy","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.723996Z","iopub.status.idle":"2022-06-28T16:42:27.724418Z","shell.execute_reply.started":"2022-06-28T16:42:27.724214Z","shell.execute_reply":"2022-06-28T16:42:27.724235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(clf, \"XGBOOST_Classifier\")","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.726193Z","iopub.status.idle":"2022-06-28T16:42:27.726578Z","shell.execute_reply.started":"2022-06-28T16:42:27.726383Z","shell.execute_reply":"2022-06-28T16:42:27.726402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****LGBM****","metadata":{}},{"cell_type":"code","source":"best_param= {\"n_estimators\":[3000],\n            \"learning_rate\":[0.01],\n            #\"lambda_l2\":24.60526923347014,\n            \"max_depth\":[16],\n            \"subsample\":[0.32],\n             \"bagging_freq\": [3],\n             #\"feature_fraction\":0.2,\n             \"random_state\": [37],\n             \"boosting_type\":['gbdt'],\n             \"min_child_samples\": [2000],\n             'objective': ['binary']\n\n            }","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.727744Z","iopub.status.idle":"2022-06-28T16:42:27.728192Z","shell.execute_reply.started":"2022-06-28T16:42:27.728001Z","shell.execute_reply":"2022-06-28T16:42:27.72802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier, early_stopping\n\nmodel = LGBMClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.730041Z","iopub.status.idle":"2022-06-28T16:42:27.730417Z","shell.execute_reply.started":"2022-06-28T16:42:27.730224Z","shell.execute_reply":"2022-06-28T16:42:27.730243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = GridSearchCV(model, best_param, n_jobs=5, \n                   cv=StratifiedKFold(n_splits=2, shuffle=True), \n                   scoring='roc_auc',\n                   verbose=2, refit=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.731458Z","iopub.status.idle":"2022-06-28T16:42:27.731848Z","shell.execute_reply.started":"2022-06-28T16:42:27.731635Z","shell.execute_reply":"2022-06-28T16:42:27.731654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.fit(X_train,y_train)\ny_pred1 = clf.predict(X_test)\ny_pred_prob1 = clf.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.736869Z","iopub.status.idle":"2022-06-28T16:42:27.737272Z","shell.execute_reply.started":"2022-06-28T16:42:27.737107Z","shell.execute_reply":"2022-06-28T16:42:27.737124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns=[\"target\"])\ny_pred1 = pd.DataFrame(y_pred1, columns=[\"prediction\"])\ny_pred_prob1 = pd.DataFrame(y_pred_prob1, columns=[\"prediction\"])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.738679Z","iopub.status.idle":"2022-06-28T16:42:27.739009Z","shell.execute_reply.started":"2022-06-28T16:42:27.738863Z","shell.execute_reply":"2022-06-28T16:42:27.738879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric_official(y_test, y_pred_prob1))","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.740606Z","iopub.status.idle":"2022-06-28T16:42:27.74096Z","shell.execute_reply.started":"2022-06-28T16:42:27.740806Z","shell.execute_reply":"2022-06-28T16:42:27.740823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\naccuracy = metrics.accuracy_score(y_test[\"target\"], y_pred1[\"prediction\"])\naccuracy","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.741714Z","iopub.status.idle":"2022-06-28T16:42:27.742049Z","shell.execute_reply.started":"2022-06-28T16:42:27.741903Z","shell.execute_reply":"2022-06-28T16:42:27.741918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(models, \"XGBOOST_Classifier\")","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.743096Z","iopub.status.idle":"2022-06-28T16:42:27.743469Z","shell.execute_reply.started":"2022-06-28T16:42:27.743302Z","shell.execute_reply":"2022-06-28T16:42:27.743327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from optuna.integration import LightGBMPruningCallback\n\ndef objective(trial, X, y):\n    param_grid = {\n        # \"device_type\": trial.suggest_categorical(\"device_type\", ['gpu']),\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [10000]),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 3000, step=20),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 200, 10000, step=100),\n        \"lambda_l1\": trial.suggest_int(\"lambda_l1\", 0, 100, step=5),\n        \"lambda_l2\": trial.suggest_int(\"lambda_l2\", 0, 100, step=5),\n        \"min_gain_to_split\": trial.suggest_float(\"min_gain_to_split\", 0, 15),\n        \"bagging_fraction\": trial.suggest_float(\n            \"bagging_fraction\", 0.2, 0.95, step=0.1\n        ),\n        \"bagging_freq\": trial.suggest_categorical(\"bagging_freq\", [1]),\n        \"feature_fraction\": trial.suggest_float(\n            \"feature_fraction\", 0.2, 0.95, step=0.1\n        ),\n    }\n\n    cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=1121218)\n\n    cv_scores = np.empty(5)\n    for idx, (train_idx, test_idx) in enumerate(cv.split(X, y)):\n        X_train, X_test = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_test = y[train_idx], y[test_idx]\n\n        model = lgbm.LGBMClassifier(objective=\"binary\", **param_grid)\n        model.fit(\n            X_train,\n            y_train,\n            eval_set=[(X_test, y_test)],\n            eval_metric=\"binary_logloss\",\n            early_stopping_rounds=100,\n            callbacks=[\n                LightGBMPruningCallback(trial, \"binary_logloss\")\n            ],  # Add a pruning callback\n        )\n        preds = model.predict_proba(X_test)\n        cv_scores[idx] = log_loss(y_test, preds)\n\n    return np.mean(cv_scores)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.745264Z","iopub.status.idle":"2022-06-28T16:42:27.745671Z","shell.execute_reply.started":"2022-06-28T16:42:27.745487Z","shell.execute_reply":"2022-06-28T16:42:27.745506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\nstudy = optuna.create_study(direction=\"minimize\", study_name=\"LGBM Classifier\")\nfunc = lambda trial: objective(trial, X, y)\nstudy.optimize(func, n_trials=20)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.746956Z","iopub.status.idle":"2022-06-28T16:42:27.747312Z","shell.execute_reply.started":"2022-06-28T16:42:27.747139Z","shell.execute_reply":"2022-06-28T16:42:27.747156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(f\"\\tBest value (rmse): {study.best_value:.5f}\")\nprint(f\"\\tBest params:\")\n\nfor key, value in study.best_params.items():\n    print(f\"\\t\\t{key}: {value}\")","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.748132Z","iopub.status.idle":"2022-06-28T16:42:27.748462Z","shell.execute_reply.started":"2022-06-28T16:42:27.748293Z","shell.execute_reply":"2022-06-28T16:42:27.74831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Ada Boost Classifier****","metadata":{}},{"cell_type":"code","source":"ada_model = AdaBoostClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.749637Z","iopub.status.idle":"2022-06-28T16:42:27.749992Z","shell.execute_reply.started":"2022-06-28T16:42:27.749821Z","shell.execute_reply":"2022-06-28T16:42:27.749838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameters = {'n_estimators': [50, 100, 150],\n             'learning_rate': [1.0, 1.5,2.0]}\n\ncv = GridSearchCV(ada_model, param_grid=parameters)","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.750943Z","iopub.status.idle":"2022-06-28T16:42:27.75128Z","shell.execute_reply.started":"2022-06-28T16:42:27.751108Z","shell.execute_reply":"2022-06-28T16:42:27.751125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ada_model.fit(X_train,y_train)\ny_pred = ada_model.predict(X_test)\ny_pred_prob = ada_model.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.752212Z","iopub.status.idle":"2022-06-28T16:42:27.752547Z","shell.execute_reply.started":"2022-06-28T16:42:27.752377Z","shell.execute_reply":"2022-06-28T16:42:27.752396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns=[\"target\"])\ny_pred = pd.DataFrame(y_pred, columns=[\"prediction\"])\ny_pred_prob = pd.DataFrame(y_pred_prob, columns=[\"prediction\"])","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.753478Z","iopub.status.idle":"2022-06-28T16:42:27.753845Z","shell.execute_reply.started":"2022-06-28T16:42:27.753639Z","shell.execute_reply":"2022-06-28T16:42:27.753655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric_official(y_test, y_pred_prob))","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.755313Z","iopub.status.idle":"2022-06-28T16:42:27.755645Z","shell.execute_reply.started":"2022-06-28T16:42:27.755477Z","shell.execute_reply":"2022-06-28T16:42:27.755494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-06-28T16:42:27.757104Z","iopub.status.idle":"2022-06-28T16:42:27.757447Z","shell.execute_reply.started":"2022-06-28T16:42:27.757274Z","shell.execute_reply":"2022-06-28T16:42:27.757292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}