{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-27T16:29:12.792138Z","iopub.execute_input":"2023-02-27T16:29:12.792564Z","iopub.status.idle":"2023-02-27T16:29:12.80398Z","shell.execute_reply.started":"2023-02-27T16:29:12.792524Z","shell.execute_reply":"2023-02-27T16:29:12.802694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows = 200000)\n#df_test = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', nrows = 200000)\ndf_train_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', nrows = 200000)\n#df_submission = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:29:20.356691Z","iopub.execute_input":"2023-02-27T16:29:20.357124Z","iopub.status.idle":"2023-02-27T16:29:27.977464Z","shell.execute_reply.started":"2023-02-27T16:29:20.357088Z","shell.execute_reply":"2023-02-27T16:29:27.975712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.merge(df_train,df_train_labels,how='inner',on=['customer_ID'])","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:29:50.730622Z","iopub.execute_input":"2023-02-27T16:29:50.731069Z","iopub.status.idle":"2023-02-27T16:29:51.344407Z","shell.execute_reply.started":"2023-02-27T16:29:50.731029Z","shell.execute_reply":"2023-02-27T16:29:51.343155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:29:59.600394Z","iopub.execute_input":"2023-02-27T16:29:59.600856Z","iopub.status.idle":"2023-02-27T16:29:59.629787Z","shell.execute_reply.started":"2023-02-27T16:29:59.600807Z","shell.execute_reply":"2023-02-27T16:29:59.628291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.tail()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:30:04.457459Z","iopub.execute_input":"2023-02-27T16:30:04.457872Z","iopub.status.idle":"2023-02-27T16:30:04.487294Z","shell.execute_reply.started":"2023-02-27T16:30:04.457838Z","shell.execute_reply":"2023-02-27T16:30:04.485898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:47:48.555633Z","iopub.execute_input":"2023-02-27T16:47:48.556313Z","iopub.status.idle":"2023-02-27T16:47:48.563811Z","shell.execute_reply.started":"2023-02-27T16:47:48.556272Z","shell.execute_reply":"2023-02-27T16:47:48.562306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.columns)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:51:06.345093Z","iopub.execute_input":"2023-02-27T16:51:06.345712Z","iopub.status.idle":"2023-02-27T16:51:06.352395Z","shell.execute_reply.started":"2023-02-27T16:51:06.345661Z","shell.execute_reply":"2023-02-27T16:51:06.350943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:48:58.26556Z","iopub.execute_input":"2023-02-27T16:48:58.266409Z","iopub.status.idle":"2023-02-27T16:49:02.592714Z","shell.execute_reply.started":"2023-02-27T16:48:58.266333Z","shell.execute_reply":"2023-02-27T16:49:02.591232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T16:50:31.9666Z","iopub.execute_input":"2023-02-27T16:50:31.967007Z","iopub.status.idle":"2023-02-27T16:50:32.099144Z","shell.execute_reply.started":"2023-02-27T16:50:31.966974Z","shell.execute_reply":"2023-02-27T16:50:32.097867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_cols = [i for i in df_train.columns if df_train[i].isnull().any()]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:08:51.098216Z","iopub.execute_input":"2023-02-27T17:08:51.099043Z","iopub.status.idle":"2023-02-27T17:08:51.211541Z","shell.execute_reply.started":"2023-02-27T17:08:51.098997Z","shell.execute_reply":"2023-02-27T17:08:51.210103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_cols","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:08:54.216786Z","iopub.execute_input":"2023-02-27T17:08:54.217206Z","iopub.status.idle":"2023-02-27T17:08:54.227688Z","shell.execute_reply.started":"2023-02-27T17:08:54.217166Z","shell.execute_reply":"2023-02-27T17:08:54.226547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(nan_cols)\n# Out of 191 column 120 have null value","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:10:27.987964Z","iopub.execute_input":"2023-02-27T17:10:27.989129Z","iopub.status.idle":"2023-02-27T17:10:27.996228Z","shell.execute_reply.started":"2023-02-27T17:10:27.989077Z","shell.execute_reply":"2023-02-27T17:10:27.9948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:10:54.177661Z","iopub.execute_input":"2023-02-27T17:10:54.178077Z","iopub.status.idle":"2023-02-27T17:10:54.198657Z","shell.execute_reply.started":"2023-02-27T17:10:54.178041Z","shell.execute_reply":"2023-02-27T17:10:54.197302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:11:53.407962Z","iopub.execute_input":"2023-02-27T17:11:53.408403Z","iopub.status.idle":"2023-02-27T17:11:56.461617Z","shell.execute_reply.started":"2023-02-27T17:11:53.408353Z","shell.execute_reply":"2023-02-27T17:11:56.460314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:13:25.627484Z","iopub.execute_input":"2023-02-27T17:13:25.627945Z","iopub.status.idle":"2023-02-27T17:13:28.094153Z","shell.execute_reply.started":"2023-02-27T17:13:25.627905Z","shell.execute_reply":"2023-02-27T17:13:28.092938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Defining categorical column to explore data visualization which is already define in \n#project explanation\n\ncategorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63',\n                    'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:38:25.62087Z","iopub.execute_input":"2023-02-27T17:38:25.62128Z","iopub.status.idle":"2023-02-27T17:38:25.627819Z","shell.execute_reply.started":"2023-02-27T17:38:25.621245Z","shell.execute_reply":"2023-02-27T17:38:25.625701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:31:53.964327Z","iopub.execute_input":"2023-02-27T18:31:53.965207Z","iopub.status.idle":"2023-02-27T18:31:53.970536Z","shell.execute_reply.started":"2023-02-27T18:31:53.965163Z","shell.execute_reply":"2023-02-27T18:31:53.969197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_cat = df_train[ ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126',\n                          'D_63','D_64', 'D_66', 'D_68']]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:44:37.527115Z","iopub.execute_input":"2023-02-27T17:44:37.527555Z","iopub.status.idle":"2023-02-27T17:44:37.548656Z","shell.execute_reply.started":"2023-02-27T17:44:37.527518Z","shell.execute_reply":"2023-02-27T17:44:37.547497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train_cat.columns:\n    print(i)\n    print(df_train_cat[i].unique())\n    print(df_train_cat[i].value_counts())\n    print('\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:59:54.815067Z","iopub.execute_input":"2023-02-27T17:59:54.815688Z","iopub.status.idle":"2023-02-27T17:59:54.925338Z","shell.execute_reply.started":"2023-02-27T17:59:54.815625Z","shell.execute_reply":"2023-02-27T17:59:54.923627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:25:57.864913Z","iopub.execute_input":"2023-02-27T18:25:57.866109Z","iopub.status.idle":"2023-02-27T18:25:57.97908Z","shell.execute_reply.started":"2023-02-27T18:25:57.866061Z","shell.execute_reply":"2023-02-27T18:25:57.97761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_Delinquency = [c for c in df_train.columns if c.startswith('D_')]\nfeat_Spend = [c for c in df_train.columns if c.startswith('S_')]\nfeat_Payment = [c for c in df_train.columns if c.startswith('P_')]\nfeat_Balance = [c for c in df_train.columns if c.startswith('B_')]\nfeat_Risk = [c for c in df_train.columns if c.startswith('R_')]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:27:37.793453Z","iopub.execute_input":"2023-02-27T18:27:37.794732Z","iopub.status.idle":"2023-02-27T18:27:37.802449Z","shell.execute_reply.started":"2023-02-27T18:27:37.794682Z","shell.execute_reply":"2023-02-27T18:27:37.800938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Total number of Delinquency variables: {len(feat_Delinquency)}')\nprint(f'Total number of Spend variables: {len(feat_Spend)}')\nprint(f'Total number of Payment variables: {len(feat_Payment)}')\nprint(f'Total number of Balance variables: {len(feat_Balance)}')\nprint(f'Total number of Risk variables: {len(feat_Risk)}')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:29:35.504831Z","iopub.execute_input":"2023-02-27T18:29:35.505263Z","iopub.status.idle":"2023-02-27T18:29:35.5137Z","shell.execute_reply.started":"2023-02-27T18:29:35.505229Z","shell.execute_reply":"2023-02-27T18:29:35.511706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=['Delinquency', 'Spend','Payment','Balance','Risk']\nvalues= [len(feat_Delinquency), len(feat_Spend),len(feat_Payment), \n         len(feat_Balance),len(feat_Risk)]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:30:28.424676Z","iopub.execute_input":"2023-02-27T18:30:28.425158Z","iopub.status.idle":"2023-02-27T18:30:28.432612Z","shell.execute_reply.started":"2023-02-27T18:30:28.425118Z","shell.execute_reply":"2023-02-27T18:30:28.430795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nimport plotly.express as px\nfrom itertools import cycle","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:31:46.885218Z","iopub.execute_input":"2023-02-27T18:31:46.885653Z","iopub.status.idle":"2023-02-27T18:31:46.891291Z","shell.execute_reply.started":"2023-02-27T18:31:46.885614Z","shell.execute_reply":"2023-02-27T18:31:46.889839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig_1 = go.Figure()\nfig_1.add_trace(go.Pie(values = values,labels = labels,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig_1.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#70d6ff\",\"#ff9770\"]),\n                  title = dict(text = 'Feature Distribution'))  \nfig_1.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:30:43.194985Z","iopub.execute_input":"2023-02-27T18:30:43.195513Z","iopub.status.idle":"2023-02-27T18:30:43.42168Z","shell.execute_reply.started":"2023-02-27T18:30:43.195409Z","shell.execute_reply":"2023-02-27T18:30:43.420501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['target'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:31:02.624055Z","iopub.execute_input":"2023-02-27T18:31:02.624637Z","iopub.status.idle":"2023-02-27T18:31:02.634302Z","shell.execute_reply.started":"2023-02-27T18:31:02.624596Z","shell.execute_reply":"2023-02-27T18:31:02.632999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:31:13.70434Z","iopub.execute_input":"2023-02-27T18:31:13.705764Z","iopub.status.idle":"2023-02-27T18:31:13.715723Z","shell.execute_reply.started":"2023-02-27T18:31:13.705711Z","shell.execute_reply":"2023-02-27T18:31:13.714121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.countplot(df_train['target'], data = df_train, palette = 'hls')\nplt.xticks(rotation = 90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:32:44.382412Z","iopub.execute_input":"2023-02-27T18:32:44.382832Z","iopub.status.idle":"2023-02-27T18:32:44.412311Z","shell.execute_reply.started":"2023-02-27T18:32:44.382797Z","shell.execute_reply":"2023-02-27T18:32:44.410405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,8))\ndf_train['target'].value_counts().plot(kind = 'pie',autopct='%1.1f%%', startangle=90)\nplt.xticks(rotation = 90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:32:36.073549Z","iopub.execute_input":"2023-02-27T18:32:36.07469Z","iopub.status.idle":"2023-02-27T18:32:36.433675Z","shell.execute_reply.started":"2023-02-27T18:32:36.074644Z","shell.execute_reply":"2023-02-27T18:32:36.4324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_class = pd.DataFrame({'count': df_train.target.value_counts(),\n            'percentage': df_train['target'].value_counts() / df_train.shape[0] * 100})","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:33:39.822828Z","iopub.execute_input":"2023-02-27T18:33:39.8238Z","iopub.status.idle":"2023-02-27T18:33:39.831652Z","shell.execute_reply.started":"2023-02-27T18:33:39.823758Z","shell.execute_reply":"2023-02-27T18:33:39.830565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_class","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:33:53.784756Z","iopub.execute_input":"2023-02-27T18:33:53.785644Z","iopub.status.idle":"2023-02-27T18:33:53.796574Z","shell.execute_reply.started":"2023-02-27T18:33:53.785567Z","shell.execute_reply":"2023-02-27T18:33:53.795309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace(go.Pie(values = target_class['count'],labels = target_class.index,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#90cf8e\",\"#ff70a6\"]),\n                  title = dict(text = 'Target Distribution'))  \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:34:10.641418Z","iopub.execute_input":"2023-02-27T18:34:10.641854Z","iopub.status.idle":"2023-02-27T18:34:10.658902Z","shell.execute_reply.started":"2023-02-27T18:34:10.641815Z","shell.execute_reply":"2023-02-27T18:34:10.657505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stat_plot = df_train.reset_index().groupby('S_2')['customer_ID'].nunique().reset_index()\nfig = go.Figure()\nfig.add_trace(go.Scatter(x = stat_plot['S_2'], y = stat_plot['customer_ID']))\nfig.update_layout(title=\"Customer Statements\", width = 800, height = 600,xaxis_title ='Statement Date',\n                  paper_bgcolor='rgb(0,0,0,0)',plot_bgcolor='rgb(0,0,0,0)') \nfig['data'][0]['line']['color']=\"#ff9770\"\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:34:25.384489Z","iopub.execute_input":"2023-02-27T18:34:25.385614Z","iopub.status.idle":"2023-02-27T18:34:25.504856Z","shell.execute_reply.started":"2023-02-27T18:34:25.385567Z","shell.execute_reply":"2023-02-27T18:34:25.503159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:34:44.68416Z","iopub.execute_input":"2023-02-27T18:34:44.685579Z","iopub.status.idle":"2023-02-27T18:34:44.691134Z","shell.execute_reply.started":"2023-02-27T18:34:44.685507Z","shell.execute_reply":"2023-02-27T18:34:44.690158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:35:11.033067Z","iopub.execute_input":"2023-02-27T18:35:11.034221Z","iopub.status.idle":"2023-02-27T18:35:11.186172Z","shell.execute_reply.started":"2023-02-27T18:35:11.034176Z","shell.execute_reply":"2023-02-27T18:35:11.184737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop('S_2', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:35:33.304684Z","iopub.execute_input":"2023-02-27T18:35:33.30513Z","iopub.status.idle":"2023-02-27T18:35:33.333717Z","shell.execute_reply.started":"2023-02-27T18:35:33.30509Z","shell.execute_reply":"2023-02-27T18:35:33.332259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_cols = [c for c in df_train.columns if (c.startswith(('D','t'))) & (c not in categorical_cols)]\ndf_del = df_train[del_cols]\nspd_cols = [c for c in df_train.columns if (c.startswith(('S','t'))) & (c not in categorical_cols)]\ndf_spd = df_train[spd_cols]\npay_cols = [c for c in df_train.columns if (c.startswith(('P','t'))) & (c not in categorical_cols)]\ndf_pay = df_train[pay_cols]\nbal_cols = [c for c in df_train.columns if (c.startswith(('B','t'))) & (c not in categorical_cols)]\ndf_bal = df_train[bal_cols]\nris_cols = [c for c in df_train.columns if (c.startswith(('R','t'))) & (c not in categorical_cols)]\ndf_ris = df_train[ris_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:37:22.838224Z","iopub.execute_input":"2023-02-27T18:37:22.838719Z","iopub.status.idle":"2023-02-27T18:37:22.865742Z","shell.execute_reply.started":"2023-02-27T18:37:22.838678Z","shell.execute_reply":"2023-02-27T18:37:22.864233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(29, 3, figsize = (35,150))\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(del_cols) - 1:\n        sns.kdeplot(x = del_cols[i], hue='target', data = df_del, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nfig.suptitle('Distribution of Delinquency Variables', fontsize = 35, x = 0.5, y = 1)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:37:53.996128Z","iopub.execute_input":"2023-02-27T18:37:53.996539Z","iopub.status.idle":"2023-02-27T18:38:25.353981Z","shell.execute_reply.started":"2023-02-27T18:37:53.996505Z","shell.execute_reply":"2023-02-27T18:38:25.352234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize =(11,11))\ncorr = df_del.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, \n            linewidths =.6 ,cmap=[\"Red\",\"Green\",\"black\"])\nplt.title('Correlation of Delinquency Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:43:31.375799Z","iopub.execute_input":"2023-02-27T18:43:31.376569Z","iopub.status.idle":"2023-02-27T18:43:32.696856Z","shell.execute_reply.started":"2023-02-27T18:43:31.376529Z","shell.execute_reply":"2023-02-27T18:43:32.695351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(8, 3, figsize = (16,18))\nfig.suptitle('Distribution of Spend Variables', fontsize = 15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(spd_cols) - 1:\n        sns.kdeplot(x = spd_cols[i], hue ='target', data = df_spd, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:46:15.999901Z","iopub.execute_input":"2023-02-27T18:46:16.000711Z","iopub.status.idle":"2023-02-27T18:46:22.929671Z","shell.execute_reply.started":"2023-02-27T18:46:16.000658Z","shell.execute_reply":"2023-02-27T18:46:22.928494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"S_cols = [c for c in df_train.columns if (c.startswith(('S')))]\ndf_S = df_train[S_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:46:33.788786Z","iopub.execute_input":"2023-02-27T18:46:33.789929Z","iopub.status.idle":"2023-02-27T18:46:33.799585Z","shell.execute_reply.started":"2023-02-27T18:46:33.789874Z","shell.execute_reply":"2023-02-27T18:46:33.798143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\ncorr = df_S.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths = .6, \n            cmap=[\"Red\",\"Green\",\"black\"])\nplt.title('Correlation of Spend Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:47:09.816822Z","iopub.execute_input":"2023-02-27T18:47:09.81811Z","iopub.status.idle":"2023-02-27T18:47:10.464921Z","shell.execute_reply.started":"2023-02-27T18:47:09.81806Z","shell.execute_reply":"2023-02-27T18:47:10.463526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize = (12,4))\nfig.suptitle('Distribution of Payment Variables',fontsize = 15)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(pay_cols) - 1:\n        sns.kdeplot(x = pay_cols[i], hue ='target', data = df_pay, fill = True, ax = ax, \n                    palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:55:58.738866Z","iopub.execute_input":"2023-02-27T18:55:58.739304Z","iopub.status.idle":"2023-02-27T18:56:00.347443Z","shell.execute_reply.started":"2023-02-27T18:55:58.739269Z","shell.execute_reply":"2023-02-27T18:56:00.345962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"P_cols = [c for c in df_train.columns if (c.startswith(('P')))]\ndf_P = df_train[P_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:56:15.025479Z","iopub.execute_input":"2023-02-27T18:56:15.026186Z","iopub.status.idle":"2023-02-27T18:56:15.03524Z","shell.execute_reply.started":"2023-02-27T18:56:15.026145Z","shell.execute_reply":"2023-02-27T18:56:15.033418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (6,6))\ncorr = df_P.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths = .6, \n            cmap=[\"Red\",\"Green\",\"black\"])\nplt.title('Correlation of Payment Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:56:48.959058Z","iopub.execute_input":"2023-02-27T18:56:48.96004Z","iopub.status.idle":"2023-02-27T18:56:49.235309Z","shell.execute_reply.started":"2023-02-27T18:56:48.959994Z","shell.execute_reply":"2023-02-27T18:56:49.234044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 4, figsize = (15,24))\nfig.suptitle('Distribution of Balance Variables',fontsize = 15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(bal_cols) - 1:\n        sns.kdeplot(x = bal_cols[i], hue ='target', data = df_bal, fill = True, ax = ax, \n                    palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:57:16.108968Z","iopub.execute_input":"2023-02-27T18:57:16.109584Z","iopub.status.idle":"2023-02-27T18:57:28.026895Z","shell.execute_reply.started":"2023-02-27T18:57:16.109546Z","shell.execute_reply":"2023-02-27T18:57:28.025838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_cols = [c for c in df_train.columns if (c.startswith(('B')))]\ndf_B = df_train[B_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:57:38.337512Z","iopub.execute_input":"2023-02-27T18:57:38.338158Z","iopub.status.idle":"2023-02-27T18:57:38.346127Z","shell.execute_reply.started":"2023-02-27T18:57:38.338116Z","shell.execute_reply":"2023-02-27T18:57:38.345109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\ncorr = df_B.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust=True, center = 0,square = True, linewidths =.6,\n            cmap=[\"Red\",\"Green\",\"black\"])\nplt.title('Correlation of Balance Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:58:09.17734Z","iopub.execute_input":"2023-02-27T18:58:09.178042Z","iopub.status.idle":"2023-02-27T18:58:10.083852Z","shell.execute_reply.started":"2023-02-27T18:58:09.177992Z","shell.execute_reply":"2023-02-27T18:58:10.082442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 3, figsize = (18,23))\nfig.suptitle('Distribution of Risk Variables',fontsize=15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(ris_cols) - 1:\n        sns.kdeplot(x = ris_cols[i], hue ='target', data = df_ris, fill = True, ax = ax,\n                    palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:58:29.913319Z","iopub.execute_input":"2023-02-27T18:58:29.913756Z","iopub.status.idle":"2023-02-27T18:58:39.485743Z","shell.execute_reply.started":"2023-02-27T18:58:29.91372Z","shell.execute_reply":"2023-02-27T18:58:39.484248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"R_cols = [c for c in df_train.columns if (c.startswith(('R')))]\ndf_R = df_train[R_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:58:47.558504Z","iopub.execute_input":"2023-02-27T18:58:47.558889Z","iopub.status.idle":"2023-02-27T18:58:47.567743Z","shell.execute_reply.started":"2023-02-27T18:58:47.558857Z","shell.execute_reply":"2023-02-27T18:58:47.566145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(11,11))\ncorr = df_R.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0, square = True, linewidths =.6, \n            cmap =[\"Red\",\"Green\",\"black\"])\nplt.title('Correlation of Risk Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:59:16.817359Z","iopub.execute_input":"2023-02-27T18:59:16.817799Z","iopub.status.idle":"2023-02-27T18:59:17.504097Z","shell.execute_reply.started":"2023-02-27T18:59:16.817763Z","shell.execute_reply":"2023-02-27T18:59:17.503106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"palette = cycle([\"#ffd670\",\"#70d6ff\",\"#ff4d6d\",\"#8338ec\",\"#90cf8e\"])\ntarg = df_train.corrwith(df_train['target'], axis=0)\nval = [str(round(v ,1) *100) + '%' for v in targ.values]\nfig = go.Figure()\nfig.add_trace(go.Bar(y=targ.index, x= targ.values, orientation='h',text = val, marker_color = next(palette)))\nfig.update_layout(title = \"Correlation of variables with Target\",width = 750, height = 3500,\n                  paper_bgcolor='rgb(0,0,0,0)',plot_bgcolor='rgb(0,0,0,0)')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T18:59:31.127244Z","iopub.execute_input":"2023-02-27T18:59:31.128506Z","iopub.status.idle":"2023-02-27T18:59:31.296965Z","shell.execute_reply.started":"2023-02-27T18:59:31.128422Z","shell.execute_reply":"2023-02-27T18:59:31.29571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:00:00.038354Z","iopub.execute_input":"2023-02-27T19:00:00.038757Z","iopub.status.idle":"2023-02-27T19:00:00.411149Z","shell.execute_reply.started":"2023-02-27T19:00:00.038723Z","shell.execute_reply":"2023-02-27T19:00:00.409664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlab_enc = LabelEncoder()\nfor cat_feat in categorical_cols:\n    df_train[cat_feat] = lab_enc.fit_transform(df_train[cat_feat])","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:00:55.631253Z","iopub.execute_input":"2023-02-27T19:00:55.631993Z","iopub.status.idle":"2023-02-27T19:00:55.716144Z","shell.execute_reply.started":"2023-02-27T19:00:55.631947Z","shell.execute_reply":"2023-02-27T19:00:55.71466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop('target', axis=1)\ny = df_train['target']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:01:06.6986Z","iopub.execute_input":"2023-02-27T19:01:06.699014Z","iopub.status.idle":"2023-02-27T19:01:06.713467Z","shell.execute_reply.started":"2023-02-27T19:01:06.69898Z","shell.execute_reply":"2023-02-27T19:01:06.712354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df_train.drop('target', axis=1)\ny= df_train['target']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:02:47.389214Z","iopub.execute_input":"2023-02-27T19:02:47.389638Z","iopub.status.idle":"2023-02-27T19:02:47.404439Z","shell.execute_reply.started":"2023-02-27T19:02:47.389602Z","shell.execute_reply":"2023-02-27T19:02:47.402547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:02:51.005715Z","iopub.execute_input":"2023-02-27T19:02:51.006106Z","iopub.status.idle":"2023-02-27T19:02:51.028179Z","shell.execute_reply.started":"2023-02-27T19:02:51.006071Z","shell.execute_reply":"2023-02-27T19:02:51.026622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:02:56.463718Z","iopub.execute_input":"2023-02-27T19:02:56.464109Z","iopub.status.idle":"2023-02-27T19:02:56.474078Z","shell.execute_reply.started":"2023-02-27T19:02:56.464077Z","shell.execute_reply":"2023-02-27T19:02:56.472337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:03:01.517702Z","iopub.execute_input":"2023-02-27T19:03:01.518092Z","iopub.status.idle":"2023-02-27T19:03:01.528221Z","shell.execute_reply.started":"2023-02-27T19:03:01.518059Z","shell.execute_reply":"2023-02-27T19:03:01.526513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# creating dataset split for prediction\nX_train, X_test , y_train , y_test = train_test_split(X,y,test_size=0.2,random_state=42) # 80-20 split\n\n# Checking split \nprint('X_train:', X_train.shape)\nprint('y_train:', y_train.shape)\nprint('X_test:', X_test.shape)\nprint('y_test:', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:03:16.298307Z","iopub.execute_input":"2023-02-27T19:03:16.298772Z","iopub.status.idle":"2023-02-27T19:03:16.405945Z","shell.execute_reply.started":"2023-02-27T19:03:16.298733Z","shell.execute_reply":"2023-02-27T19:03:16.404497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nclf = CatBoostClassifier(iterations = 3000, random_state = 42)\nclf.fit(X_train, y_train, eval_set = [(X_test, y_test)], cat_features=categorical_cols,  verbose = 100)\npreds = clf.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:04:01.366744Z","iopub.execute_input":"2023-02-27T19:04:01.367158Z","iopub.status.idle":"2023-02-27T19:07:09.177144Z","shell.execute_reply.started":"2023-02-27T19:04:01.367125Z","shell.execute_reply":"2023-02-27T19:07:09.175598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:07:09.179841Z","iopub.execute_input":"2023-02-27T19:07:09.18125Z","iopub.status.idle":"2023-02-27T19:07:09.207333Z","shell.execute_reply.started":"2023-02-27T19:07:09.181201Z","shell.execute_reply":"2023-02-27T19:07:09.205815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:07:09.208857Z","iopub.execute_input":"2023-02-27T19:07:09.209341Z","iopub.status.idle":"2023-02-27T19:07:09.214863Z","shell.execute_reply.started":"2023-02-27T19:07:09.209306Z","shell.execute_reply":"2023-02-27T19:07:09.213464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:07:09.21705Z","iopub.execute_input":"2023-02-27T19:07:09.217621Z","iopub.status.idle":"2023-02-27T19:07:09.235176Z","shell.execute_reply.started":"2023-02-27T19:07:09.217585Z","shell.execute_reply":"2023-02-27T19:07:09.233396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix  \ncm = confusion_matrix(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:08:59.36915Z","iopub.execute_input":"2023-02-27T19:08:59.370556Z","iopub.status.idle":"2023-02-27T19:08:59.379421Z","shell.execute_reply.started":"2023-02-27T19:08:59.3705Z","shell.execute_reply":"2023-02-27T19:08:59.37812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:09:06.340862Z","iopub.execute_input":"2023-02-27T19:09:06.341297Z","iopub.status.idle":"2023-02-27T19:09:06.349261Z","shell.execute_reply.started":"2023-02-27T19:09:06.341259Z","shell.execute_reply":"2023-02-27T19:09:06.348009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:09:17.389883Z","iopub.execute_input":"2023-02-27T19:09:17.391051Z","iopub.status.idle":"2023-02-27T19:09:17.642926Z","shell.execute_reply.started":"2023-02-27T19:09:17.391006Z","shell.execute_reply":"2023-02-27T19:09:17.641044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:09:33.269738Z","iopub.execute_input":"2023-02-27T19:09:33.270132Z","iopub.status.idle":"2023-02-27T19:09:33.286369Z","shell.execute_reply.started":"2023-02-27T19:09:33.270099Z","shell.execute_reply":"2023-02-27T19:09:33.285238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.8\ndf_train = df_train.drop(df_train.columns[df_train.isnull().mean() >= threshold], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:09:51.089695Z","iopub.execute_input":"2023-02-27T19:09:51.090124Z","iopub.status.idle":"2023-02-27T19:09:51.116437Z","shell.execute_reply.started":"2023-02-27T19:09:51.090088Z","shell.execute_reply":"2023-02-27T19:09:51.114338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:10:31.53644Z","iopub.execute_input":"2023-02-27T19:10:31.536916Z","iopub.status.idle":"2023-02-27T19:10:31.56704Z","shell.execute_reply.started":"2023-02-27T19:10:31.536875Z","shell.execute_reply":"2023-02-27T19:10:31.565644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:10:45.144717Z","iopub.execute_input":"2023-02-27T19:10:45.145939Z","iopub.status.idle":"2023-02-27T19:10:45.152639Z","shell.execute_reply.started":"2023-02-27T19:10:45.145894Z","shell.execute_reply":"2023-02-27T19:10:45.151691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:10:58.206955Z","iopub.execute_input":"2023-02-27T19:10:58.207744Z","iopub.status.idle":"2023-02-27T19:10:58.215641Z","shell.execute_reply.started":"2023-02-27T19:10:58.207702Z","shell.execute_reply":"2023-02-27T19:10:58.21431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:11:14.784964Z","iopub.execute_input":"2023-02-27T19:11:14.785398Z","iopub.status.idle":"2023-02-27T19:11:15.03686Z","shell.execute_reply.started":"2023-02-27T19:11:14.785347Z","shell.execute_reply":"2023-02-27T19:11:15.03473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:11:25.806941Z","iopub.execute_input":"2023-02-27T19:11:25.807408Z","iopub.status.idle":"2023-02-27T19:11:25.825687Z","shell.execute_reply.started":"2023-02-27T19:11:25.807346Z","shell.execute_reply":"2023-02-27T19:11:25.823884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:11:59.68634Z","iopub.execute_input":"2023-02-27T19:11:59.68755Z","iopub.status.idle":"2023-02-27T19:11:59.705085Z","shell.execute_reply.started":"2023-02-27T19:11:59.687486Z","shell.execute_reply":"2023-02-27T19:11:59.703727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop('target', axis=1)\ny = df_train['target']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:12:03.669708Z","iopub.execute_input":"2023-02-27T19:12:03.670928Z","iopub.status.idle":"2023-02-27T19:12:03.685586Z","shell.execute_reply.started":"2023-02-27T19:12:03.670866Z","shell.execute_reply":"2023-02-27T19:12:03.683895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:12:14.53747Z","iopub.execute_input":"2023-02-27T19:12:14.537879Z","iopub.status.idle":"2023-02-27T19:12:14.546178Z","shell.execute_reply.started":"2023-02-27T19:12:14.537844Z","shell.execute_reply":"2023-02-27T19:12:14.54476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesClassifier\nimport matplotlib.pyplot as plt\nmodel = ExtraTreesClassifier()\nmodel.fit(X,y)\nprint(model.feature_importances_)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:12:30.346949Z","iopub.execute_input":"2023-02-27T19:12:30.347399Z","iopub.status.idle":"2023-02-27T19:12:35.719601Z","shell.execute_reply.started":"2023-02-27T19:12:30.347342Z","shell.execute_reply":"2023-02-27T19:12:35.717974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.iloc[:,:-1]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:12:44.826747Z","iopub.execute_input":"2023-02-27T19:12:44.827149Z","iopub.status.idle":"2023-02-27T19:12:44.84548Z","shell.execute_reply.started":"2023-02-27T19:12:44.827115Z","shell.execute_reply":"2023-02-27T19:12:44.844072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_importances = pd.Series(model.feature_importances_, index=X.columns)\nfeat_importances.nlargest(10).plot(kind='barh')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:12:58.833121Z","iopub.execute_input":"2023-02-27T19:12:58.833581Z","iopub.status.idle":"2023-02-27T19:12:59.025343Z","shell.execute_reply.started":"2023-02-27T19:12:58.833541Z","shell.execute_reply":"2023-02-27T19:12:59.023827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test , y_train , y_test = train_test_split(X,y,test_size=0.2,random_state=42) # 80-20 split\nprint('X_train:', X_train.shape)\nprint('y_train:', y_train.shape)\nprint('X_test:', X_test.shape)\nprint('y_test:', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:13:14.496945Z","iopub.execute_input":"2023-02-27T19:13:14.497403Z","iopub.status.idle":"2023-02-27T19:13:14.525842Z","shell.execute_reply.started":"2023-02-27T19:13:14.497349Z","shell.execute_reply":"2023-02-27T19:13:14.524598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression  \nlog_r= LogisticRegression(random_state=0)  \nlog_r.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:13:31.844496Z","iopub.execute_input":"2023-02-27T19:13:31.844941Z","iopub.status.idle":"2023-02-27T19:13:32.393341Z","shell.execute_reply.started":"2023-02-27T19:13:31.844902Z","shell.execute_reply":"2023-02-27T19:13:32.391754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_lr= log_r.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:13:52.43418Z","iopub.execute_input":"2023-02-27T19:13:52.434633Z","iopub.status.idle":"2023-02-27T19:13:52.446356Z","shell.execute_reply.started":"2023-02-27T19:13:52.434595Z","shell.execute_reply":"2023-02-27T19:13:52.444838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test, y_pred_lr)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:14:01.195133Z","iopub.execute_input":"2023-02-27T19:14:01.195621Z","iopub.status.idle":"2023-02-27T19:14:01.206111Z","shell.execute_reply.started":"2023-02-27T19:14:01.195581Z","shell.execute_reply":"2023-02-27T19:14:01.204058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred_lr)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:14:13.117008Z","iopub.execute_input":"2023-02-27T19:14:13.117501Z","iopub.status.idle":"2023-02-27T19:14:13.125437Z","shell.execute_reply.started":"2023-02-27T19:14:13.117458Z","shell.execute_reply":"2023-02-27T19:14:13.124058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:14:24.51681Z","iopub.execute_input":"2023-02-27T19:14:24.517274Z","iopub.status.idle":"2023-02-27T19:14:24.525073Z","shell.execute_reply.started":"2023-02-27T19:14:24.517234Z","shell.execute_reply":"2023-02-27T19:14:24.52377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:14:39.755773Z","iopub.execute_input":"2023-02-27T19:14:39.756439Z","iopub.status.idle":"2023-02-27T19:14:40.043828Z","shell.execute_reply.started":"2023-02-27T19:14:39.7564Z","shell.execute_reply":"2023-02-27T19:14:40.042523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred_lr))","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:14:53.996934Z","iopub.execute_input":"2023-02-27T19:14:53.997719Z","iopub.status.idle":"2023-02-27T19:14:54.014745Z","shell.execute_reply.started":"2023-02-27T19:14:53.997661Z","shell.execute_reply":"2023-02-27T19:14:54.012892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndt = DecisionTreeClassifier()\ndt.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:15:04.859025Z","iopub.execute_input":"2023-02-27T19:15:04.859817Z","iopub.status.idle":"2023-02-27T19:15:10.273312Z","shell.execute_reply.started":"2023-02-27T19:15:04.859767Z","shell.execute_reply":"2023-02-27T19:15:10.271939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_dt= dt.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:15:19.076794Z","iopub.execute_input":"2023-02-27T19:15:19.077207Z","iopub.status.idle":"2023-02-27T19:15:19.087085Z","shell.execute_reply.started":"2023-02-27T19:15:19.07717Z","shell.execute_reply":"2023-02-27T19:15:19.085948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test, y_pred_dt)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:15:29.525299Z","iopub.execute_input":"2023-02-27T19:15:29.525718Z","iopub.status.idle":"2023-02-27T19:15:29.53394Z","shell.execute_reply.started":"2023-02-27T19:15:29.525684Z","shell.execute_reply":"2023-02-27T19:15:29.532443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred_dt)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:15:39.63646Z","iopub.execute_input":"2023-02-27T19:15:39.63696Z","iopub.status.idle":"2023-02-27T19:15:39.645783Z","shell.execute_reply.started":"2023-02-27T19:15:39.636921Z","shell.execute_reply":"2023-02-27T19:15:39.643915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:15:51.746762Z","iopub.execute_input":"2023-02-27T19:15:51.747182Z","iopub.status.idle":"2023-02-27T19:15:51.754012Z","shell.execute_reply.started":"2023-02-27T19:15:51.747146Z","shell.execute_reply":"2023-02-27T19:15:51.753152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:16:01.116811Z","iopub.execute_input":"2023-02-27T19:16:01.117224Z","iopub.status.idle":"2023-02-27T19:16:01.39582Z","shell.execute_reply.started":"2023-02-27T19:16:01.117191Z","shell.execute_reply":"2023-02-27T19:16:01.394494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred_dt))","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:16:12.476671Z","iopub.execute_input":"2023-02-27T19:16:12.478179Z","iopub.status.idle":"2023-02-27T19:16:12.498417Z","shell.execute_reply.started":"2023-02-27T19:16:12.478123Z","shell.execute_reply":"2023-02-27T19:16:12.497087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier  \nrf_c = RandomForestClassifier(n_estimators= 10, criterion=\"entropy\")  \nrf_c.fit(X_train, y_train) ","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:16:24.527266Z","iopub.execute_input":"2023-02-27T19:16:24.528077Z","iopub.status.idle":"2023-02-27T19:16:26.157975Z","shell.execute_reply.started":"2023-02-27T19:16:24.528035Z","shell.execute_reply":"2023-02-27T19:16:26.156729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_rf_c= rf_c.predict(X_test)  ","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:16:46.846971Z","iopub.execute_input":"2023-02-27T19:16:46.847661Z","iopub.status.idle":"2023-02-27T19:16:46.871915Z","shell.execute_reply.started":"2023-02-27T19:16:46.847611Z","shell.execute_reply":"2023-02-27T19:16:46.870698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test, y_pred_rf_c)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:16:55.505072Z","iopub.execute_input":"2023-02-27T19:16:55.505712Z","iopub.status.idle":"2023-02-27T19:16:55.515986Z","shell.execute_reply.started":"2023-02-27T19:16:55.505667Z","shell.execute_reply":"2023-02-27T19:16:55.514649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm= confusion_matrix(y_test, y_pred_rf_c) ","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:17:11.315951Z","iopub.execute_input":"2023-02-27T19:17:11.3164Z","iopub.status.idle":"2023-02-27T19:17:11.324542Z","shell.execute_reply.started":"2023-02-27T19:17:11.316348Z","shell.execute_reply":"2023-02-27T19:17:11.323341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:17:17.356697Z","iopub.execute_input":"2023-02-27T19:17:17.357089Z","iopub.status.idle":"2023-02-27T19:17:17.365025Z","shell.execute_reply.started":"2023-02-27T19:17:17.357055Z","shell.execute_reply":"2023-02-27T19:17:17.363744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:17:27.70058Z","iopub.execute_input":"2023-02-27T19:17:27.700985Z","iopub.status.idle":"2023-02-27T19:17:27.988336Z","shell.execute_reply.started":"2023-02-27T19:17:27.700952Z","shell.execute_reply":"2023-02-27T19:17:27.98687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred_rf_c))","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:17:40.955124Z","iopub.execute_input":"2023-02-27T19:17:40.956686Z","iopub.status.idle":"2023-02-27T19:17:40.973089Z","shell.execute_reply.started":"2023-02-27T19:17:40.956623Z","shell.execute_reply":"2023-02-27T19:17:40.971792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:17:54.392082Z","iopub.execute_input":"2023-02-27T19:17:54.392662Z","iopub.status.idle":"2023-02-27T19:17:54.398374Z","shell.execute_reply.started":"2023-02-27T19:17:54.392595Z","shell.execute_reply":"2023-02-27T19:17:54.397404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds = StratifiedKFold(n_splits = 5, shuffle = True, random_state = 40)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:18:10.301057Z","iopub.execute_input":"2023-02-27T19:18:10.301531Z","iopub.status.idle":"2023-02-27T19:18:10.307039Z","shell.execute_reply.started":"2023-02-27T19:18:10.301485Z","shell.execute_reply":"2023-02-27T19:18:10.305654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def grid_search(model,folds,params,scoring):\n    grid_search = GridSearchCV(model,\n                                cv=folds, \n                                param_grid=params, \n                                scoring=scoring, \n                                n_jobs=-1, verbose=1)\n    return grid_search","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:18:20.62045Z","iopub.execute_input":"2023-02-27T19:18:20.620865Z","iopub.status.idle":"2023-02-27T19:18:20.62889Z","shell.execute_reply.started":"2023-02-27T19:18:20.620831Z","shell.execute_reply":"2023-02-27T19:18:20.627307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_best_score_params(model):\n    print(\"Best Score: \", model.best_score_)\n    print(\"Best Hyperparameters: \", model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:18:31.28031Z","iopub.execute_input":"2023-02-27T19:18:31.280824Z","iopub.status.idle":"2023-02-27T19:18:31.287251Z","shell.execute_reply.started":"2023-02-27T19:18:31.280785Z","shell.execute_reply":"2023-02-27T19:18:31.286052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg = LogisticRegression()\nlog_params = {'C': [0.01, 1, 10], \n          'penalty': ['l1', 'l2'],\n          'solver': ['liblinear','newton-cg','saga']\n         }\ngrid_search_log = grid_search(log_reg, folds, log_params, scoring=None)\ngrid_search_log.fit(X_train, y_train)\nprint_best_score_params(grid_search_log)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:18:41.700802Z","iopub.execute_input":"2023-02-27T19:18:41.701834Z","iopub.status.idle":"2023-02-27T19:20:27.723498Z","shell.execute_reply.started":"2023-02-27T19:18:41.70178Z","shell.execute_reply":"2023-02-27T19:20:27.72172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense\n\nmodel = Sequential()\nmodel.add(Dense(64, input_dim=X_train.shape[1], activation='relu'))\nmodel.add(Dense(32, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:22:00.502693Z","iopub.execute_input":"2023-02-27T19:22:00.503238Z","iopub.status.idle":"2023-02-27T19:22:10.799232Z","shell.execute_reply.started":"2023-02-27T19:22:00.503187Z","shell.execute_reply":"2023-02-27T19:22:10.797773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_train, y_train, epochs=20, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:22:17.814571Z","iopub.execute_input":"2023-02-27T19:22:17.815699Z","iopub.status.idle":"2023-02-27T19:22:40.823148Z","shell.execute_reply.started":"2023-02-27T19:22:17.815639Z","shell.execute_reply":"2023-02-27T19:22:40.821782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(X_test, y_test)\nprint('Test accuracy:', test_acc)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:23:32.852313Z","iopub.execute_input":"2023-02-27T19:23:32.853718Z","iopub.status.idle":"2023-02-27T19:23:33.160239Z","shell.execute_reply.started":"2023-02-27T19:23:32.853664Z","shell.execute_reply":"2023-02-27T19:23:33.15883Z"},"trusted":true},"execution_count":null,"outputs":[]}]}