{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-16T08:42:16.57902Z","iopub.execute_input":"2022-07-16T08:42:16.579544Z","iopub.status.idle":"2022-07-16T08:42:16.619631Z","shell.execute_reply.started":"2022-07-16T08:42:16.57944Z","shell.execute_reply":"2022-07-16T08:42:16.618215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:16.684693Z","iopub.execute_input":"2022-07-16T08:42:16.685139Z","iopub.status.idle":"2022-07-16T08:42:16.690141Z","shell.execute_reply.started":"2022-07-16T08:42:16.685096Z","shell.execute_reply":"2022-07-16T08:42:16.6888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:16.7822Z","iopub.execute_input":"2022-07-16T08:42:16.783493Z","iopub.status.idle":"2022-07-16T08:42:16.792808Z","shell.execute_reply.started":"2022-07-16T08:42:16.783438Z","shell.execute_reply":"2022-07-16T08:42:16.791767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Libraries for data visulaization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore') #To supress warnings","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:16.853254Z","iopub.execute_input":"2022-07-16T08:42:16.854009Z","iopub.status.idle":"2022-07-16T08:42:18.04161Z","shell.execute_reply.started":"2022-07-16T08:42:16.853967Z","shell.execute_reply":"2022-07-16T08:42:18.040103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Loading train_labels.csv file","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.044195Z","iopub.execute_input":"2022-07-16T08:42:18.044592Z","iopub.status.idle":"2022-07-16T08:42:18.055497Z","shell.execute_reply.started":"2022-07-16T08:42:18.044553Z","shell.execute_reply":"2022-07-16T08:42:18.050819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\") #Loading dataset\ntrain_labels.head() #To see first five rows of the dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.057006Z","iopub.status.idle":"2022-07-16T08:42:18.057922Z","shell.execute_reply.started":"2022-07-16T08:42:18.057607Z","shell.execute_reply":"2022-07-16T08:42:18.057662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['target'].value_counts() #Counts unique values","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.059869Z","iopub.status.idle":"2022-07-16T08:42:18.060903Z","shell.execute_reply.started":"2022-07-16T08:42:18.060599Z","shell.execute_reply":"2022-07-16T08:42:18.060628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading train_data.ftr, test_data.ftr files\nThere are a total of 190 variables in the dataset with approximately 450,000 customers in the training set and 925,000 in the test set. This dataset is prodigious and reading it directly consumes the entire memory. There are two options to overcome this problem:\n\nRead data chunk by chunk\nUse a compressed dataset\nThe disadvantage of reading data chunk by chunk is we cannot explore the entire data distribution.\n\nSecond option is using the compressed version of the train and test sets provided by @munumbutt's AMEX-Feather-Dataset.","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_feather(\"../input/amexfeather/train_data.ftr\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.062805Z","iopub.status.idle":"2022-07-16T08:42:18.063742Z","shell.execute_reply.started":"2022-07-16T08:42:18.063266Z","shell.execute_reply":"2022-07-16T08:42:18.0633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.065613Z","iopub.status.idle":"2022-07-16T08:42:18.066947Z","shell.execute_reply.started":"2022-07-16T08:42:18.066584Z","shell.execute_reply":"2022-07-16T08:42:18.066621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_feather(\"../input/amexfeather/test_data.ftr\")\nsample_submission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.06866Z","iopub.status.idle":"2022-07-16T08:42:18.069516Z","shell.execute_reply.started":"2022-07-16T08:42:18.069176Z","shell.execute_reply":"2022-07-16T08:42:18.06921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.071502Z","iopub.status.idle":"2022-07-16T08:42:18.072378Z","shell.execute_reply.started":"2022-07-16T08:42:18.072033Z","shell.execute_reply":"2022-07-16T08:42:18.072067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of train data:\",train_data.shape)\nprint(\"Shape of test data:\",test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.07429Z","iopub.status.idle":"2022-07-16T08:42:18.075135Z","shell.execute_reply.started":"2022-07-16T08:42:18.074807Z","shell.execute_reply":"2022-07-16T08:42:18.074842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To get a summary of a dataframe we use info() function. This method prints information about a dataframe including the index dtypes, column dtypes, non-null values and memory usage.","metadata":{}},{"cell_type":"code","source":"train_data.info(max_cols=191 ,show_counts = True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.077297Z","iopub.status.idle":"2022-07-16T08:42:18.078153Z","shell.execute_reply.started":"2022-07-16T08:42:18.077825Z","shell.execute_reply":"2022-07-16T08:42:18.077859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can verify the presence of null values using isnull() function.\n","metadata":{}},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.080791Z","iopub.status.idle":"2022-07-16T08:42:18.081895Z","shell.execute_reply.started":"2022-07-16T08:42:18.081518Z","shell.execute_reply":"2022-07-16T08:42:18.081555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will take the latest statement for each customer.","metadata":{}},{"cell_type":"code","source":"train_data1 = train_data.groupby('customer_ID').tail(1).set_index('customer_ID')\ntest_data1 = test_data.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.083371Z","iopub.status.idle":"2022-07-16T08:42:18.084575Z","shell.execute_reply.started":"2022-07-16T08:42:18.084216Z","shell.execute_reply":"2022-07-16T08:42:18.084259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Exploratory Data Analysis\nThe target binary variable is calculated by observing 18 months performance window after the latest credit card statement, and if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.\n\nThe dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\nD_*: Delinquency variables\nS_*: Spend variables\nP_*: Payment variables\nB_*: Balance variables\nR_*: Risk variables\nWith the following features being categorical: B_30, B_38, D_63, D_64, D_66, D_68, D_114, D_116, D_117, D_120, D_126.","metadata":{}},{"cell_type":"code","source":"feat_Delinquency = [c for c in train_data.columns if c.startswith('D_')]\nfeat_Spend = [c for c in train_data.columns if c.startswith('S_')]\nfeat_Payment = [c for c in train_data.columns if c.startswith('P_')]\nfeat_Balance = [c for c in train_data.columns if c.startswith('B_')]\nfeat_Risk = [c for c in train_data.columns if c.startswith('R_')]\nprint(f'Total number of Delinquency variables: {len(feat_Delinquency)}')\nprint(f'Total number of Spend variables: {len(feat_Spend)}')\nprint(f'Total number of Payment variables: {len(feat_Payment)}')\nprint(f'Total number of Balance variables: {len(feat_Balance)}')\nprint(f'Total number of Risk variables: {len(feat_Risk)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.086418Z","iopub.status.idle":"2022-07-16T08:42:18.087403Z","shell.execute_reply.started":"2022-07-16T08:42:18.087052Z","shell.execute_reply":"2022-07-16T08:42:18.087087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\nlabels=['Delinquency', 'Spend','Payment','Balance','Risk']\nvalues= [len(feat_Delinquency), len(feat_Spend),len(feat_Payment), len(feat_Balance),len(feat_Risk)]\nfig = px.pie(train_data, values=values, names=labels)\nfig.update_traces(textposition='inside', textinfo='percent+label')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.089443Z","iopub.status.idle":"2022-07-16T08:42:18.090946Z","shell.execute_reply.started":"2022-07-16T08:42:18.090567Z","shell.execute_reply":"2022-07-16T08:42:18.090603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA on train data","metadata":{}},{"cell_type":"code","source":"missing_train_data = train_data.isna().sum().div(len(train_data)).mul(100).sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.09273Z","iopub.status.idle":"2022-07-16T08:42:18.093892Z","shell.execute_reply.started":"2022-07-16T08:42:18.093505Z","shell.execute_reply":"2022-07-16T08:42:18.09354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(25,10))\nsns.barplot(x=missing_train_data[:100].index, y=missing_train_data[:100].values, ax=ax[0])\nsns.barplot(x=missing_train_data[100:].index, y=missing_train_data[100:].values, ax=ax[1])\nax[0].set_ylabel(\"Percentage [%]\"), ax[1].set_ylabel(\"Percentage [%]\")\nax[0].tick_params(axis='x', rotation=90); ax[1].tick_params(axis='x', rotation=90)\nplt.suptitle(\"Amount of missing data (in train data)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.095893Z","iopub.status.idle":"2022-07-16T08:42:18.096564Z","shell.execute_reply.started":"2022-07-16T08:42:18.096234Z","shell.execute_reply":"2022-07-16T08:42:18.096266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = train_data[\"target\"].value_counts()\ntarget_0 = round((target[0]/train_data['target'].count()*100),2)\ntarget_1 = round((target[1]/train_data['target'].count()*100),2)\ntarget_percentage = {'Target':['0', '1'], 'Percentage':[target_0, target_1]} \ndf_target_percentage = pd.DataFrame(target_percentage)\ngroupedvalues = df_target_percentage.groupby('Percentage').sum().reset_index()\n\nax = sns.barplot(x='Target',y='Percentage', data=df_target_percentage, errwidth=0)\nplt.title('Percentage of target_0 vs target_1 on train data')\nax.bar_label(ax.containers[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.098978Z","iopub.status.idle":"2022-07-16T08:42:18.100205Z","shell.execute_reply.started":"2022-07-16T08:42:18.099935Z","shell.execute_reply":"2022-07-16T08:42:18.099961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above plot, it can be inferred visually that target = 0 rows are more than target = 1 rows. It means the dataset is imbalanced with majority class 'target = 0' and minority 'target = 1'.\n\n24.91% of customers had a default - it is worth talking to these two different groups and find dome key differences.","metadata":{}},{"cell_type":"code","source":"print(f'Number of unique customers: {train_data[\"customer_ID\"].nunique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.101197Z","iopub.status.idle":"2022-07-16T08:42:18.102295Z","shell.execute_reply.started":"2022-07-16T08:42:18.102047Z","shell.execute_reply":"2022-07-16T08:42:18.102074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = train_data.groupby(['customer_ID','target']).size().reset_index()\ncustomers = customers.rename(columns={0:'Count'})\ncustomers.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.103569Z","iopub.status.idle":"2022-07-16T08:42:18.104222Z","shell.execute_reply.started":"2022-07-16T08:42:18.104003Z","shell.execute_reply":"2022-07-16T08:42:18.104026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1, figsize=(15,5))\nsns.histplot(x='Count', data=customers, hue='target', stat='percent', multiple=\"dodge\", bins=np.arange(0,14), ax=ax)\nax.bar_label(ax.containers[0], fmt='%.f%%')\nax.bar_label(ax.containers[1], fmt='%.f%%')\nplt.title(\"Count Distribution on train data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.105682Z","iopub.status.idle":"2022-07-16T08:42:18.106527Z","shell.execute_reply.started":"2022-07-16T08:42:18.106304Z","shell.execute_reply":"2022-07-16T08:42:18.106327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del customers","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.107901Z","iopub.status.idle":"2022-07-16T08:42:18.108288Z","shell.execute_reply.started":"2022-07-16T08:42:18.108109Z","shell.execute_reply":"2022-07-16T08:42:18.108128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"statement = train_data.groupby('customer_ID')['S_2'].max().reset_index()\nstatement.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.109486Z","iopub.status.idle":"2022-07-16T08:42:18.109923Z","shell.execute_reply.started":"2022-07-16T08:42:18.109716Z","shell.execute_reply":"2022-07-16T08:42:18.109745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(20,5))\nax = plt.axes()\n\na=sns.countplot(data=statement,x=statement['S_2'])\nx_dates = statement['S_2'].dt.strftime('%Y-%m-%d').sort_values().unique()\nax.set_xticklabels(labels=x_dates, rotation=45, ha='right')\nplt.title(\"Customer's Last Date Statement's Count Distribution\")\nplt.show()\n\ndel statement","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.110916Z","iopub.status.idle":"2022-07-16T08:42:18.111292Z","shell.execute_reply.started":"2022-07-16T08:42:18.111112Z","shell.execute_reply":"2022-07-16T08:42:18.11113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.112685Z","iopub.status.idle":"2022-07-16T08:42:18.11306Z","shell.execute_reply.started":"2022-07-16T08:42:18.112886Z","shell.execute_reply":"2022-07-16T08:42:18.112903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA on test data","metadata":{}},{"cell_type":"code","source":"missing_test_data = test_data.isna().sum().div(len(test_data)).mul(100).sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.11399Z","iopub.status.idle":"2022-07-16T08:42:18.114349Z","shell.execute_reply.started":"2022-07-16T08:42:18.114173Z","shell.execute_reply":"2022-07-16T08:42:18.11419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(25,10))\nsns.barplot(x=missing_test_data[:100].index, y=missing_test_data[:100].values, ax=ax[0])\nsns.barplot(x=missing_test_data[100:].index, y=missing_test_data[100:].values, ax=ax[1])\nax[0].set_ylabel(\"Percentage [%]\"), ax[1].set_ylabel(\"Percentage [%]\")\nax[0].tick_params(axis='x', rotation=90); ax[1].tick_params(axis='x', rotation=90)\nplt.suptitle(\"Amount of missing data (in test data)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.115489Z","iopub.status.idle":"2022-07-16T08:42:18.115923Z","shell.execute_reply.started":"2022-07-16T08:42:18.115728Z","shell.execute_reply":"2022-07-16T08:42:18.115749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = test_data.groupby(['customer_ID']).size().reset_index()\ncustomers = customers.rename(columns={0:'Count'})\ncustomers.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.116973Z","iopub.status.idle":"2022-07-16T08:42:18.117341Z","shell.execute_reply.started":"2022-07-16T08:42:18.117162Z","shell.execute_reply":"2022-07-16T08:42:18.11718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1, figsize=(15,5))\nsns.countplot(x = 'Count',data=customers)\nax.grid(linestyle=\"--\",axis='y',color='gray')\nplt.title(\"Count Distribution on test data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.118402Z","iopub.status.idle":"2022-07-16T08:42:18.118846Z","shell.execute_reply.started":"2022-07-16T08:42:18.118616Z","shell.execute_reply":"2022-07-16T08:42:18.118634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del customers","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.120243Z","iopub.status.idle":"2022-07-16T08:42:18.1206Z","shell.execute_reply.started":"2022-07-16T08:42:18.12043Z","shell.execute_reply":"2022-07-16T08:42:18.120447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"statement = test_data.groupby('customer_ID')['S_2'].max().reset_index()\nfig = plt.figure(figsize=(20,5))\nax = plt.axes()\n\na=sns.countplot(data=statement,x=statement['S_2'])\nx_dates = statement['S_2'].dt.strftime('%Y-%m-%d').sort_values().unique()\nax.set_xticklabels(labels=x_dates, rotation=45)\nplt.title(\"Customer's Last Date Statement's Count Distribution on test data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.121828Z","iopub.status.idle":"2022-07-16T08:42:18.122188Z","shell.execute_reply.started":"2022-07-16T08:42:18.122015Z","shell.execute_reply":"2022-07-16T08:42:18.122032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del statement","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.123479Z","iopub.status.idle":"2022-07-16T08:42:18.123917Z","shell.execute_reply.started":"2022-07-16T08:42:18.123701Z","shell.execute_reply":"2022-07-16T08:42:18.123744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.126079Z","iopub.status.idle":"2022-07-16T08:42:18.126465Z","shell.execute_reply.started":"2022-07-16T08:42:18.126286Z","shell.execute_reply":"2022-07-16T08:42:18.126305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distribution of Continuos Deliquency Variables**","metadata":{}},{"cell_type":"code","source":"cat_var= ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nobj_col=['customer_ID', 'S_2']","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.127985Z","iopub.status.idle":"2022-07-16T08:42:18.128374Z","shell.execute_reply.started":"2022-07-16T08:42:18.128187Z","shell.execute_reply":"2022-07-16T08:42:18.128204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in cat_var:\n    sns.countplot(data=train_data,x=i, hue='target')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.12941Z","iopub.status.idle":"2022-07-16T08:42:18.129837Z","shell.execute_reply.started":"2022-07-16T08:42:18.129603Z","shell.execute_reply":"2022-07-16T08:42:18.129621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_cols = [c for c in train_data1.columns if (c.startswith(('D','t'))) & (c not in cat_var)] #Delinquency\ndf_del = train_data1[del_cols]\nspd_cols = [c for c in train_data1.columns if (c.startswith(('S','t'))) & (c not in cat_var)] #Spend\ndf_spd = train_data1[spd_cols]\npay_cols = [c for c in train_data1.columns if (c.startswith(('P','t'))) & (c not in cat_var)] #Payment\ndf_pay = train_data1[pay_cols]\nbal_cols = [c for c in train_data1.columns if (c.startswith(('B','t'))) & (c not in cat_var)] #Balance\ndf_bal = train_data1[bal_cols]\nris_cols = [c for c in train_data1.columns if (c.startswith(('R','t'))) & (c not in cat_var)] #Risk\ndf_ris = train_data1[ris_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.130886Z","iopub.status.idle":"2022-07-16T08:42:18.13125Z","shell.execute_reply.started":"2022-07-16T08:42:18.131077Z","shell.execute_reply":"2022-07-16T08:42:18.131093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.133127Z","iopub.status.idle":"2022-07-16T08:42:18.133508Z","shell.execute_reply.started":"2022-07-16T08:42:18.133332Z","shell.execute_reply":"2022-07-16T08:42:18.133349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\ndef kdeplot(cols,df,title,figsize):\n    plt_cols = 5\n    plt_rows = math.ceil(len(cols)/plt_cols)\n    \n    fig, axes = plt.subplots(plt_rows, plt_cols, figsize = figsize)\n    for i, ax in enumerate(axes.reshape(-1)):\n        if i < len(cols) - 1:\n            sns.kdeplot(x = cols[i], hue='target', data = df, fill = True, ax = ax)\n            ax.tick_params()\n            ax.xaxis.get_label()\n            ax.set_ylabel('')\n    fig.suptitle(title, fontsize = 35, x = 0.5, y = 1)\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.135139Z","iopub.status.idle":"2022-07-16T08:42:18.135691Z","shell.execute_reply.started":"2022-07-16T08:42:18.135337Z","shell.execute_reply":"2022-07-16T08:42:18.135355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def corr(df,title,figsize):\n    plt.figure(figsize =figsize)\n    corr = df.corr()\n    mask = np.triu(np.ones_like(corr, dtype = bool))\n    sns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths =.6)\n    plt.title(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.13684Z","iopub.status.idle":"2022-07-16T08:42:18.137345Z","shell.execute_reply.started":"2022-07-16T08:42:18.137153Z","shell.execute_reply":"2022-07-16T08:42:18.137172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.13877Z","iopub.status.idle":"2022-07-16T08:42:18.139146Z","shell.execute_reply.started":"2022-07-16T08:42:18.138971Z","shell.execute_reply":"2022-07-16T08:42:18.138988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kdeplot(del_cols, df_del, 'Distribution of Delinquency Variables',(35,150))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.140415Z","iopub.status.idle":"2022-07-16T08:42:18.14086Z","shell.execute_reply.started":"2022-07-16T08:42:18.14061Z","shell.execute_reply":"2022-07-16T08:42:18.140626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.14171Z","iopub.status.idle":"2022-07-16T08:42:18.142081Z","shell.execute_reply.started":"2022-07-16T08:42:18.1419Z","shell.execute_reply":"2022-07-16T08:42:18.141917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr(df_del,'Correlation of Delinquency Variables',(20,20))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.143617Z","iopub.status.idle":"2022-07-16T08:42:18.144062Z","shell.execute_reply.started":"2022-07-16T08:42:18.143879Z","shell.execute_reply":"2022-07-16T08:42:18.143899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kdeplot(spd_cols,df_spd,'Distribution of Spend Variables',((16,16)))\ncorr(df_spd,'Correlation of Spend Variables',(11,11))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.145295Z","iopub.status.idle":"2022-07-16T08:42:18.145693Z","shell.execute_reply.started":"2022-07-16T08:42:18.14549Z","shell.execute_reply":"2022-07-16T08:42:18.145506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kdeplot(pay_cols,df_pay,'Distribution of Payment Variables',(12,4))\ncorr(df_pay,'Correlation of Payment Variables',(6,6))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.147745Z","iopub.status.idle":"2022-07-16T08:42:18.148133Z","shell.execute_reply.started":"2022-07-16T08:42:18.147951Z","shell.execute_reply":"2022-07-16T08:42:18.147969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kdeplot(bal_cols,df_bal,'Distribution of Balance Variables',(15,24))\ncorr(df_bal,'Correlation of Balance Variables',(11,11))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.156351Z","iopub.status.idle":"2022-07-16T08:42:18.156865Z","shell.execute_reply.started":"2022-07-16T08:42:18.156653Z","shell.execute_reply":"2022-07-16T08:42:18.156678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kdeplot(ris_cols,df_ris,'Distribution of Risk Variables',(18,23))\ncorr(df_ris,'Correlation of Risk Variables',(11,11))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.158301Z","iopub.status.idle":"2022-07-16T08:42:18.158785Z","shell.execute_reply.started":"2022-07-16T08:42:18.158556Z","shell.execute_reply":"2022-07-16T08:42:18.158576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\ntarget = train_data1.corrwith(train_data1['target'], axis=0)\nval = [str(round(v ,1) *100) + '%' for v in target.values]\nfig = go.Figure()\nfig.add_trace(go.Bar(y=target.index, x= target.values, orientation='h',text = val))\nfig.update_layout(title = \"Correlation of variables with Target\",width = 700, height = 3000)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.160502Z","iopub.status.idle":"2022-07-16T08:42:18.160964Z","shell.execute_reply.started":"2022-07-16T08:42:18.160761Z","shell.execute_reply":"2022-07-16T08:42:18.160782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.162554Z","iopub.status.idle":"2022-07-16T08:42:18.162989Z","shell.execute_reply.started":"2022-07-16T08:42:18.162798Z","shell.execute_reply":"2022-07-16T08:42:18.162817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\nfor cat_feat in cat_var:\n    train_data1[cat_feat] = le.fit_transform(train_data1[cat_feat])\n    test_data1[cat_feat] = le.transform(test_data1[cat_feat])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.164859Z","iopub.status.idle":"2022-07-16T08:42:18.165249Z","shell.execute_reply.started":"2022-07-16T08:42:18.165071Z","shell.execute_reply":"2022-07-16T08:42:18.165089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data1.drop(['S_2'],axis=1,inplace=True)\ntest_data1.drop(['S_2'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.166984Z","iopub.status.idle":"2022-07-16T08:42:18.167615Z","shell.execute_reply.started":"2022-07-16T08:42:18.167303Z","shell.execute_reply":"2022-07-16T08:42:18.167333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_data1.drop('target', axis=1) # Putting feature variables into X\ny = train_data1['target'] # Putting target variable to y","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.170201Z","iopub.status.idle":"2022-07-16T08:42:18.171007Z","shell.execute_reply.started":"2022-07-16T08:42:18.170543Z","shell.execute_reply":"2022-07-16T08:42:18.170575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split # Importing the train test split library\nX = train_data1.drop('target', axis=1) # Putting feature variables into X\ny = train_data1['target'] # Putting target variable to y\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, random_state=42, stratify = y) # Splitting data into train and test set 75:25","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.172633Z","iopub.status.idle":"2022-07-16T08:42:18.173132Z","shell.execute_reply.started":"2022-07-16T08:42:18.172934Z","shell.execute_reply":"2022-07-16T08:42:18.172954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.174592Z","iopub.status.idle":"2022-07-16T08:42:18.175058Z","shell.execute_reply.started":"2022-07-16T08:42:18.17486Z","shell.execute_reply":"2022-07-16T08:42:18.17488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import metrics\nfrom sklearn import metrics\nfrom sklearn.metrics import confusion_matrix # Prints the correct and also incorrect values in number count .\nfrom sklearn.metrics import f1_score #Combines precision, recall into a single metric by taking the harmonic mean\nfrom sklearn.metrics import classification_report #Used to show the precision, recall, F1 Score, and support of our trained classification model.\nfrom sklearn.metrics import roc_curve, auc, roc_auc_score #Shows the trade-off between sensitivity (or TPR) and specificity (1 – FPR).","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.177368Z","iopub.execute_input":"2022-07-16T08:42:18.177724Z","iopub.status.idle":"2022-07-16T08:42:18.385272Z","shell.execute_reply.started":"2022-07-16T08:42:18.177689Z","shell.execute_reply":"2022-07-16T08:42:18.384134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ROC** curve : ROC curve (Receiver Operating Characteristics Curve) is a metric used to measure the performance of a classifier model. The ROC curve depicts the rate of true positives (The model correctly predicts the positive class) with respect to the rate of false positives (the model predicts as positive class but in actual case it is a negative class), highlighting the sensitivity (Sensitivity is a measure of how well a machine learning model can detect positive instances. It is also known as the true positive rate (TPR) or recall of the classifier model.","metadata":{}},{"cell_type":"code","source":"# ROC Curve function\n\ndef draw_roc( actual, probs ):\n    fpr, tpr, thresholds = metrics.roc_curve( actual, probs,\n                                              drop_intermediate = False )\n    auc_score = metrics.roc_auc_score( actual, probs )\n    plt.figure(figsize=(5, 5))\n    plt.plot( fpr, tpr, label='ROC curve (area = %0.2f)' % auc_score )\n    plt.plot([0, 1], [0, 1], 'k--')\n    plt.xlim([0.0, 1.0]) # X axis limit is from 0 to 1\n    plt.ylim([0.0, 1.05]) # Y axis limit is from 0 to 1.05\n    plt.xlabel('False Positive Rate or [1 - True Negative Rate]') #The actual one is negative but predicted as positive\n    plt.ylabel('True Positive Rate') #Both actual and predicted value came out negative\n    plt.title('Receiver operating characteristic example')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n\n    return None\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.391924Z","iopub.execute_input":"2022-07-16T08:42:18.392333Z","iopub.status.idle":"2022-07-16T08:42:18.401185Z","shell.execute_reply.started":"2022-07-16T08:42:18.392298Z","shell.execute_reply":"2022-07-16T08:42:18.399942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Confusion matrix : A Confusion matrix is an n x n matrix used for evaluating the performance of a classification model. Here n is the number of target variables or target classes. The matrix compares the actual target values with those predicted by the machine learning model.","metadata":{}},{"cell_type":"code","source":"# Created a common function to plot confusion matrix\ndef Plot_confusion_matrix(y_test, pred_test):\n    cm = confusion_matrix(y_test, pred_test)\n    plt.clf()\n    plt.imshow(cm, interpolation='nearest', cmap=plt.cm.Accent)\n    categoryNames = ['Paid','Default']\n    plt.title('Confusion Matrix')\n    plt.ylabel('Actual label')\n    plt.xlabel('Predicted label')\n    ticks = np.arange(len(categoryNames))\n    plt.xticks(ticks, categoryNames, rotation=45)\n    plt.yticks(ticks, categoryNames)\n    s = [['TP','FN'], ['FP', 'TN']]\n    for i in range(2):\n        for j in range(2):\n            plt.text(j,i, str(s[i][j])+\" = \"+str(cm[i][j]),fontsize=12)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.402823Z","iopub.execute_input":"2022-07-16T08:42:18.403212Z","iopub.status.idle":"2022-07-16T08:42:18.413622Z","shell.execute_reply.started":"2022-07-16T08:42:18.403178Z","shell.execute_reply":"2022-07-16T08:42:18.412613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Precision Recall curve** : Precision-Recall is a useful measure of success of prediction when the classes are very imbalanced. In information retrieval, precision is a measure of result relevancy, while recall is a measure of how many truly relevant results are returned.\n\nThe precision-recall curve shows the tradeoff between precision and recall for different threshold. A high area under the curve represents both high recall and high precision, where high precision relates to a low false positive rate, and high recall relates to a low false negative rate. High scores for both show that the classifier is returning accurate results (high precision), as well as returning a majority of all positive results (high recall).\n\nA system with high recall but low precision returns many results, but most of its predicted labels are incorrect when compared to the training labels. A system with high precision but low recall is just the opposite, returning very few results, but most of its predicted labels are correct when compared to the training labels. An ideal system with high precision and high recall will return many results, with all results labeled correctly.","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve \ndef precision_recall_plot(y_true, y_probs, label):\n   \n    p, r, thresh = precision_recall_curve(y_true, y_probs)\n    p, r, thresh = list(p), list(r), list(thresh)\n    p.pop()\n    r.pop()\n\n    fig, axis = plt.subplots(nrows=1, ncols=1, figsize=(10, 5))\n    sns.lineplot(thresh, p, estimator=None,\n                     label='Precision', ax=axis)\n    axis.set_xlabel('Threshold')\n    axis.set_ylabel('Precision')\n    axis.legend(loc='lower left')\n    axis_twin = axis.twinx()\n    sns.lineplot(thresh, r, estimator=None,color='limegreen', label='Recall', ax=axis_twin)\n    axis_twin.set_ylabel('Recall')\n    axis_twin.set_ylim(0, 1)\n    axis_twin.legend(bbox_to_anchor=(0.24, 0.18))\n    axis.set_xlim(0, 1)\n    axis.set_ylim(0, 1)\n    axis.set_title('Precision Vs Recall')\n    \n    plt.close()\n    \n    fig.subplots_adjust(wspace=5)\n    fig.tight_layout()\n    display(fig)\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.415422Z","iopub.execute_input":"2022-07-16T08:42:18.415853Z","iopub.status.idle":"2022-07-16T08:42:18.432003Z","shell.execute_reply.started":"2022-07-16T08:42:18.415814Z","shell.execute_reply":"2022-07-16T08:42:18.430715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def metric_All(clf_name,y,y_pred):\n    accuracy_metric = metrics.accuracy_score(y, y_pred) #To check accuracy\n    f1score_metric = f1_score(y, y_pred) #To see F1score\n    confusion = metrics.confusion_matrix(y, y_pred)\n    TP = confusion[0,0] # true positive \n    TN = confusion[1,1] # true negatives\n    FP = confusion[1,0] # false positives\n    FN = confusion[0,1] # false negatives\n    sensitivity = TP / float(TP+FN) #Measure of how well a machine learning model can detect positive instances.\n    specificity =  TN / float(TN+FP) # Metric that evaluates a model's ability to predict true negatives of each available category.\n    print(\"Classification report\")\n    print(classification_report(y, y_pred))\n    print(\"ROC curve\")\n    draw_roc(y, y_pred)\n    roc_metric = metrics.roc_auc_score(y, y_pred)\n    fpr, tpr, thresholds = metrics.roc_curve(y, y_pred)\n    Plot_confusion_matrix(y, y_pred)\n    precision_recall_plot(y, y_pred,clf_name)\n    return (accuracy_metric,f1score_metric,sensitivity,specificity,roc_metric)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.434093Z","iopub.execute_input":"2022-07-16T08:42:18.434739Z","iopub.status.idle":"2022-07-16T08:42:18.445064Z","shell.execute_reply.started":"2022-07-16T08:42:18.434688Z","shell.execute_reply":"2022-07-16T08:42:18.443548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier , early_stopping , log_evaluation\nfrom catboost import CatBoostClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:18.446389Z","iopub.execute_input":"2022-07-16T08:42:18.446797Z","iopub.status.idle":"2022-07-16T08:42:19.997927Z","shell.execute_reply.started":"2022-07-16T08:42:18.446763Z","shell.execute_reply":"2022-07-16T08:42:19.996715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reference of these params took from: https://www.kaggle.com/code/kellibelcher/amex-default-prediction-eda-lgbm-baseline\nparams = {'boosting_type': 'gbdt',\n          'n_estimators': 1000,\n          'num_leaves': 50,\n          'learning_rate': 0.05,\n          'colsample_bytree': 0.9,\n          'min_child_samples': 2000,\n          'max_bins': 500,\n          'reg_alpha': 2,\n          'objective': 'binary',\n          'random_state': 21}","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:19.999603Z","iopub.execute_input":"2022-07-16T08:42:20.000152Z","iopub.status.idle":"2022-07-16T08:42:20.008227Z","shell.execute_reply.started":"2022-07-16T08:42:20.000102Z","shell.execute_reply":"2022-07-16T08:42:20.006148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifiers = {\n    \"Light GBM\": LGBMClassifier(**params),\n    \"Cat Boost Classifier\": CatBoostClassifier(iterations = 1000, random_state = 21, nan_mode ='Min')\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.010929Z","iopub.execute_input":"2022-07-16T08:42:20.011668Z","iopub.status.idle":"2022-07-16T08:42:20.025872Z","shell.execute_reply.started":"2022-07-16T08:42:20.011582Z","shell.execute_reply":"2022-07-16T08:42:20.024805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Results = pd.DataFrame(columns=['Model','Accuracy','F1 score','Sensitivity','Specificity','ROC value'])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.027405Z","iopub.execute_input":"2022-07-16T08:42:20.027839Z","iopub.status.idle":"2022-07-16T08:42:20.050118Z","shell.execute_reply.started":"2022-07-16T08:42:20.027801Z","shell.execute_reply":"2022-07-16T08:42:20.048508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.051985Z","iopub.execute_input":"2022-07-16T08:42:20.052667Z","iopub.status.idle":"2022-07-16T08:42:20.187457Z","shell.execute_reply.started":"2022-07-16T08:42:20.052537Z","shell.execute_reply":"2022-07-16T08:42:20.185785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_probs = []\nlgbm_preds = []\ncat_probs = []\ncat_preds = []","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.189073Z","iopub.status.idle":"2022-07-16T08:42:20.18953Z","shell.execute_reply.started":"2022-07-16T08:42:20.189326Z","shell.execute_reply":"2022-07-16T08:42:20.189347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model(clf, df_Results,  X_train,y_train, X_test, y_test):\n    for i, (clf_name,clf) in enumerate(classifiers.items()):\n        if clf_name == \"Light GBM\":\n            clf.fit(X_train, y_train, eval_set=[(X_train, y_train), (X_test, y_test)],\n                                           callbacks=[early_stopping(100), log_evaluation(200)],\n                                           eval_metric=['auc','binary_logloss'])\n            lgbm_prob = clf.predict_proba(X_test)[:,1] #On split test set (Split from train dataset)\n            lgbm_probs.append(lgbm_prob)\n            \n            #On the original test dataset\n            lgbm_preds.append(clf.predict_proba(test_data1)[:, 1])\n            \n        elif clf_name == \"Cat Boost Classifier\":\n            clf.fit(X_train, y_train,eval_set = [(X_test, y_test)], cat_features=cat_var,  verbose = 100)\n            cat_prob = clf.predict_proba(X_test)[:,1] #On split test set (Split from train dataset)\n            cat_probs.append(cat_prob) \n            \n            #On the original test dataset\n            cat_preds.append(clf.predict_proba(test_data1)[:, 1])\n            \n            \n        print(clf_name)\n        y_pred = clf.predict(X_test) #Predictions on the split set\n        s = metric_All(clf_name,y_test,y_pred)\n        y_test_pred = clf.predict(test_data1)   # Predictions on the original test dataset \n        df_Results = df_Results.append(pd.DataFrame({'Model': clf_name+ ' on test data','Accuracy': s[0] ,'F1 score':s[1],'Sensitivity':s[2],'Specificity':s[3],'ROC value': s[4]}, index=[0]),ignore_index= True)\n        print('-'*60 )\n        gc.collect()\n    return df_Results","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.192369Z","iopub.status.idle":"2022-07-16T08:42:20.192852Z","shell.execute_reply.started":"2022-07-16T08:42:20.192606Z","shell.execute_reply":"2022-07-16T08:42:20.192626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Results = model(classifiers,df_Results, X_train,y_train, X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.194051Z","iopub.status.idle":"2022-07-16T08:42:20.194506Z","shell.execute_reply.started":"2022-07-16T08:42:20.194299Z","shell.execute_reply":"2022-07-16T08:42:20.194322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(np.mean(lgbm_preds, axis=0).tolist())","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.196464Z","iopub.status.idle":"2022-07-16T08:42:20.19695Z","shell.execute_reply.started":"2022-07-16T08:42:20.196733Z","shell.execute_reply":"2022-07-16T08:42:20.196755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(np.mean(cat_preds, axis=0).tolist())","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.198338Z","iopub.status.idle":"2022-07-16T08:42:20.198816Z","shell.execute_reply.started":"2022-07-16T08:42:20.198583Z","shell.execute_reply":"2022-07-16T08:42:20.198604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Results","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.20161Z","iopub.status.idle":"2022-07-16T08:42:20.202129Z","shell.execute_reply.started":"2022-07-16T08:42:20.201917Z","shell.execute_reply":"2022-07-16T08:42:20.201941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above, it is clear that both the classifiers performed well. In this notebook, I am submitting LGBM predictions. The same can be done with Cat Boost predictions as well.","metadata":{}},{"cell_type":"markdown","source":"# SUBMISSION****","metadata":{}},{"cell_type":"code","source":"sample_submission['prediction']=np.mean(lgbm_preds, axis=0)\ndf=pd.DataFrame(data={'Target':sample_submission['prediction'].apply(lambda x:1 if x > 0.5 else 0)})\ndf=df.Target.value_counts(normalize=True)\ndf.rename(index={0:'Paid', 1:'Default'},inplace=True)\nfig=go.Figure()\nfig.add_trace(go.Pie(labels=df.index, values=df*100, \n                     showlegend=True, hovertemplate = \"%{label} Accounts: %{value:.2f}%<extra></extra>\"))\nfig.update_layout(title='Predicted Target Distribution', \n                  legend=dict(traceorder='reversed',y=1,x=1),\n                  uniformtext_minsize=15, uniformtext_mode='hide',width=700)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.203425Z","iopub.status.idle":"2022-07-16T08:42:20.203964Z","shell.execute_reply.started":"2022-07-16T08:42:20.203736Z","shell.execute_reply":"2022-07-16T08:42:20.20376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index=False)\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T08:42:20.205304Z","iopub.status.idle":"2022-07-16T08:42:20.205781Z","shell.execute_reply.started":"2022-07-16T08:42:20.205528Z","shell.execute_reply":"2022-07-16T08:42:20.205548Z"},"trusted":true},"execution_count":null,"outputs":[]}]}