{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.display import Image\nImage(filename='../input/amex-images/Surviving-a-Credit-Card-Default.png', width=\"1100\", height='50')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-10T06:58:31.893048Z","iopub.execute_input":"2022-11-10T06:58:31.893410Z","iopub.status.idle":"2022-11-10T06:58:31.912402Z","shell.execute_reply.started":"2022-11-10T06:58:31.893378Z","shell.execute_reply":"2022-11-10T06:58:31.911319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# **<span style=\"color:#87CEEB;\">Problem Statement</span>**\n\n- In this competition, you’ll apply your machine learning skills to predict credit default. Specifically, you will leverage an industrial scale data set to build a machine learning model that challenges the current model in production.\n\n# **<span style=\"color:#87CEEB;\">Data</span>**\n\nThe objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile. The target binary variable is calculated by observing 18 months performance window after the latest credit card statement, and if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.\n\nThe dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\n- D_* = Delinquency variables\n- S_* = Spend variables\n- P_* = Payment variables\n- B_* = Balance variables\n- R_* = Risk variables\n\nwith the following features being categorical:\n\n`['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']`\n\nYour task is to predict, for each customer_ID, the probability of a future payment default (target = 1).\n\nNote that the negative class has been subsampled for this dataset at 5%, and thus receives a 20x weighting in the scoring metric\n\n- `train_data.csv` - training data with multiple statement dates per customer_ID\n- `train_labels.csv - target label for each customer_ID\n- `test_data.csv` - corresponding test data; your objective is to predict the target label for each customer_ID\n- `sample_submission.csv` - a sample submission file in the correct format","metadata":{}},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Importing libraries</span>**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import auc, roc_curve\nfrom matplotlib import style\n\nplt.style.use('Solarize_Light2')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-13T15:58:34.228427Z","iopub.execute_input":"2022-11-13T15:58:34.229361Z","iopub.status.idle":"2022-11-13T15:58:36.577527Z","shell.execute_reply.started":"2022-11-13T15:58:34.229258Z","shell.execute_reply":"2022-11-13T15:58:36.576307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_feather('../input/amex-default-prediction-feather/train.feather')\ntest_df = pd.read_feather('../input/amex-default-prediction-feather/test.feather')\ntrain_labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\ntrain_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:00:14.983042Z","iopub.execute_input":"2022-11-13T16:00:14.983718Z","iopub.status.idle":"2022-11-13T16:01:21.870637Z","shell.execute_reply.started":"2022-11-13T16:00:14.983679Z","shell.execute_reply":"2022-11-13T16:01:21.869573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns',None)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:01:21.872604Z","iopub.execute_input":"2022-11-13T16:01:21.873249Z","iopub.status.idle":"2022-11-13T16:01:21.878573Z","shell.execute_reply.started":"2022-11-13T16:01:21.873206Z","shell.execute_reply":"2022-11-13T16:01:21.877407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Data Merging</span>**","metadata":{}},{"cell_type":"code","source":"#%time train_df = train_df.merge(train_labels, on='customer_ID', how='left')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:23:48.861110Z","iopub.execute_input":"2022-11-13T15:23:48.861491Z","iopub.status.idle":"2022-11-13T15:25:13.702138Z","shell.execute_reply.started":"2022-11-13T15:23:48.861459Z","shell.execute_reply":"2022-11-13T15:25:13.700929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=train_df.iloc[:200000]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:01:58.882012Z","iopub.execute_input":"2022-11-13T16:01:58.883155Z","iopub.status.idle":"2022-11-13T16:01:58.887777Z","shell.execute_reply.started":"2022-11-13T16:01:58.883114Z","shell.execute_reply":"2022-11-13T16:01:58.886564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:02:11.969053Z","iopub.execute_input":"2022-11-13T16:02:11.969812Z","iopub.status.idle":"2022-11-13T16:02:12.151341Z","shell.execute_reply.started":"2022-11-13T16:02:11.969774Z","shell.execute_reply":"2022-11-13T16:02:12.150261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp=train_labels[\"target\"].value_counts(normalize=True)*100","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:30:28.308035Z","iopub.execute_input":"2022-11-13T15:30:28.308416Z","iopub.status.idle":"2022-11-13T15:30:28.324386Z","shell.execute_reply.started":"2022-11-13T15:30:28.308384Z","shell.execute_reply":"2022-11-13T15:30:28.323314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,5))\nax = sns.barplot(x=tmp.index, y=tmp.values)\nax.bar_label(ax.containers[0], fmt='%.f%%')\nplt.title(\"Distribution of a target variable\")\nplt.ylabel(\"Percentage [%]\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:30:47.152288Z","iopub.execute_input":"2022-11-13T15:30:47.152995Z","iopub.status.idle":"2022-11-13T15:30:47.394184Z","shell.execute_reply.started":"2022-11-13T15:30:47.152959Z","shell.execute_reply":"2022-11-13T15:30:47.393149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"deliq=[i for i in train_df.columns if i.startswith(\"D\")]\nspend=[i for i in train_df.columns if i.startswith(\"S\")]\npayment=[i for i in train_df.columns if i.startswith(\"P\")]\nbalance=[i for i in train_df.columns if i.startswith(\"B\")]\nrisk=[i for i in train_df.columns if i.startswith(\"R\")]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:30:53.300726Z","iopub.execute_input":"2022-11-13T15:30:53.301097Z","iopub.status.idle":"2022-11-13T15:30:53.309558Z","shell.execute_reply.started":"2022-11-13T15:30:53.301058Z","shell.execute_reply":"2022-11-13T15:30:53.308283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[deliq].describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:30:57.606590Z","iopub.execute_input":"2022-11-13T15:30:57.606968Z","iopub.status.idle":"2022-11-13T15:31:00.500837Z","shell.execute_reply.started":"2022-11-13T15:30:57.606933Z","shell.execute_reply":"2022-11-13T15:31:00.499798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"D_* = Delinquency variables\nS_* = Spend variables\nP_* = Payment variables\nB_* = Balance variables\nR_* = Risk variables","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:39:02.583721Z","iopub.execute_input":"2022-11-10T09:39:02.584105Z","iopub.status.idle":"2022-11-10T09:39:02.593542Z","shell.execute_reply.started":"2022-11-10T09:39:02.584073Z","shell.execute_reply":"2022-11-10T09:39:02.592102Z"}}},{"cell_type":"code","source":"cat_col=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:02:22.388115Z","iopub.execute_input":"2022-11-13T16:02:22.388537Z","iopub.status.idle":"2022-11-13T16:02:22.394485Z","shell.execute_reply.started":"2022-11-13T16:02:22.388503Z","shell.execute_reply":"2022-11-13T16:02:22.393230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[cat_col]=train_df[cat_col].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:31:08.191808Z","iopub.execute_input":"2022-11-13T15:31:08.192165Z","iopub.status.idle":"2022-11-13T15:31:08.230239Z","shell.execute_reply.started":"2022-11-13T15:31:08.192133Z","shell.execute_reply":"2022-11-13T15:31:08.229303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=train_df['B_30'].dropna().unique()\norder=train_df['B_30'].value_counts().index\nplt.figure(figsize=(14,14))\nplt.subplot(1, 2, 2)\nplt.title('Pie Chart', fontweight='bold', fontsize=14)\nplt.pie(train_df['B_30'].value_counts(), labels=order, pctdistance=0.67, autopct='%.2f%%',\n        wedgeprops=dict(alpha=0.8), textprops={'fontsize':12})\ncentre=plt.Circle((0, 0), 0.45, fc='white')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:31:13.732076Z","iopub.execute_input":"2022-11-13T15:31:13.732462Z","iopub.status.idle":"2022-11-13T15:31:13.897972Z","shell.execute_reply.started":"2022-11-13T15:31:13.732427Z","shell.execute_reply":"2022-11-13T15:31:13.896687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = [c for c in list(train_df.columns) if c not in ['customer_ID','S_2']]\ncat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\nnum_features = [col for col in all_cols if col not in cat_features]\n\nnum_df = train_df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\nnum_df.columns = ['_'.join(x) for x in num_df.columns]\n\ncat_df=train_df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\ncat_df.columns = ['_'.join(x) for x in cat_df.columns]\nmain_df=pd.concat([num_df,cat_df], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:06:43.326800Z","iopub.execute_input":"2022-11-13T16:06:43.327223Z","iopub.status.idle":"2022-11-13T16:06:46.653509Z","shell.execute_reply.started":"2022-11-13T16:06:43.327169Z","shell.execute_reply":"2022-11-13T16:06:46.652336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_df","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:06:49.837466Z","iopub.execute_input":"2022-11-13T16:06:49.837831Z","iopub.status.idle":"2022-11-13T16:06:50.730472Z","shell.execute_reply.started":"2022-11-13T16:06:49.837798Z","shell.execute_reply":"2022-11-13T16:06:50.729279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time train_df = main_df.merge(train_labels, on='customer_ID', how='left')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:07:44.592590Z","iopub.execute_input":"2022-11-13T16:07:44.593022Z","iopub.status.idle":"2022-11-13T16:07:44.894141Z","shell.execute_reply.started":"2022-11-13T16:07:44.592987Z","shell.execute_reply":"2022-11-13T16:07:44.892891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:07:49.221788Z","iopub.execute_input":"2022-11-13T16:07:49.222215Z","iopub.status.idle":"2022-11-13T16:07:49.234737Z","shell.execute_reply.started":"2022-11-13T16:07:49.222157Z","shell.execute_reply":"2022-11-13T16:07:49.233558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"customer_ID\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:08:44.571791Z","iopub.execute_input":"2022-11-13T16:08:44.572167Z","iopub.status.idle":"2022-11-13T16:08:44.590625Z","shell.execute_reply.started":"2022-11-13T16:08:44.572132Z","shell.execute_reply":"2022-11-13T16:08:44.589453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust=pd.DataFrame(train_df[\"customer_ID\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:28:59.352313Z","iopub.execute_input":"2022-11-10T09:28:59.352719Z","iopub.status.idle":"2022-11-10T09:28:59.372660Z","shell.execute_reply.started":"2022-11-10T09:28:59.352686Z","shell.execute_reply":"2022-11-10T09:28:59.371637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:31:39.285535Z","iopub.execute_input":"2022-11-10T09:31:39.286269Z","iopub.status.idle":"2022-11-10T09:31:39.319868Z","shell.execute_reply.started":"2022-11-10T09:31:39.286230Z","shell.execute_reply":"2022-11-10T09:31:39.318822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust=train_df[train_df[\"customer_ID\"]==\"0000099d6bd597052cdcda90ffabf56573fe9d7c79be5fbac11a8ed792feb62a\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:29:20.942279Z","iopub.execute_input":"2022-11-10T09:29:20.942656Z","iopub.status.idle":"2022-11-10T09:29:20.971107Z","shell.execute_reply.started":"2022-11-10T09:29:20.942620Z","shell.execute_reply":"2022-11-10T09:29:20.969811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:29:23.678023Z","iopub.execute_input":"2022-11-10T09:29:23.678810Z","iopub.status.idle":"2022-11-10T09:29:23.687568Z","shell.execute_reply.started":"2022-11-10T09:29:23.678770Z","shell.execute_reply":"2022-11-10T09:29:23.686415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust2=train_df[train_df[\"customer_ID\"]==\"05ef792b3ee85dd86e9716fdc0ca548c4ccaa80d21f06faf3e68ac3da4307746\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:30:12.387450Z","iopub.execute_input":"2022-11-10T09:30:12.388206Z","iopub.status.idle":"2022-11-10T09:30:12.407423Z","shell.execute_reply.started":"2022-11-10T09:30:12.388166Z","shell.execute_reply":"2022-11-10T09:30:12.406223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust3=train_df[train_df[\"customer_ID\"]==\"05ef72d6f964d7af444ff77815851ad82cdd804d9295cd52c92054754548d12f\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:36:31.496358Z","iopub.execute_input":"2022-11-10T09:36:31.496754Z","iopub.status.idle":"2022-11-10T09:36:31.515316Z","shell.execute_reply.started":"2022-11-10T09:36:31.496716Z","shell.execute_reply":"2022-11-10T09:36:31.514282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust3[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T09:37:15.706316Z","iopub.execute_input":"2022-11-10T09:37:15.706717Z","iopub.status.idle":"2022-11-10T09:37:15.716621Z","shell.execute_reply.started":"2022-11-10T09:37:15.706683Z","shell.execute_reply":"2022-11-10T09:37:15.715616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(45,45))\nmat = train_df[deliq].corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T09:40:38.749315Z","iopub.execute_input":"2022-11-11T09:40:38.749698Z","iopub.status.idle":"2022-11-11T09:40:57.440964Z","shell.execute_reply.started":"2022-11-11T09:40:38.749664Z","shell.execute_reply":"2022-11-11T09:40:57.438877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(45,45))\nmat = train_df[spend].corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T09:40:57.442824Z","iopub.execute_input":"2022-11-11T09:40:57.443419Z","iopub.status.idle":"2022-11-11T09:41:00.053929Z","shell.execute_reply.started":"2022-11-11T09:40:57.443379Z","shell.execute_reply":"2022-11-11T09:41:00.052979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(10,10))\nmat = train_df[payment].corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T09:41:17.058316Z","iopub.execute_input":"2022-11-11T09:41:17.058698Z","iopub.status.idle":"2022-11-11T09:41:17.326979Z","shell.execute_reply.started":"2022-11-11T09:41:17.058663Z","shell.execute_reply":"2022-11-11T09:41:17.326026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(45,45))\nmat = train_df[risk].corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T09:41:23.584278Z","iopub.execute_input":"2022-11-11T09:41:23.584645Z","iopub.status.idle":"2022-11-11T09:41:27.021050Z","shell.execute_reply.started":"2022-11-11T09:41:23.584614Z","shell.execute_reply":"2022-11-11T09:41:27.020171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(45,45))\nmat = train_df[balance].corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-11T09:41:32.179023Z","iopub.execute_input":"2022-11-11T09:41:32.179416Z","iopub.status.idle":"2022-11-11T09:41:36.441337Z","shell.execute_reply.started":"2022-11-11T09:41:32.179380Z","shell.execute_reply":"2022-11-11T09:41:36.440279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Checking Missing values</span>**","metadata":{}},{"cell_type":"code","source":"for i in train_df.columns:\n    print(f\"{i}--->{((train_df[i].isnull().sum())/200000)*100}%\")","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:08:55.939306Z","iopub.execute_input":"2022-11-13T16:08:55.939681Z","iopub.status.idle":"2022-11-13T16:08:56.207659Z","shell.execute_reply.started":"2022-11-13T16:08:55.939652Z","shell.execute_reply":"2022-11-13T16:08:56.206381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan = pd.DataFrame(train_df.isna().sum(), columns = ['NaN_sum'])\nnan['feat'] = nan.index\nnan['Perc(%)'] = (nan['NaN_sum']/200000)*100\nnan = nan[nan['NaN_sum'] > 0]\nnan = nan.sort_values(by = ['NaN_sum'])\nnan['Usability'] = np.where(nan['Perc(%)'] > 70, 'Discard', 'Keep')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:09.744796Z","iopub.execute_input":"2022-11-13T16:10:09.745198Z","iopub.status.idle":"2022-11-13T16:10:09.855362Z","shell.execute_reply.started":"2022-11-13T16:10:09.745147Z","shell.execute_reply":"2022-11-13T16:10:09.854461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:16.671637Z","iopub.execute_input":"2022-11-13T16:10:16.671996Z","iopub.status.idle":"2022-11-13T16:10:17.677666Z","shell.execute_reply.started":"2022-11-13T16:10:16.671965Z","shell.execute_reply":"2022-11-13T16:10:17.676856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Data to Plot</span>**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (45,20))\nsns.barplot(x = nan['feat'], y = nan['Perc(%)'])\nplt.xticks(rotation=45)\nplt.title('Features containing Nan')\nplt.xlabel('Features')\nplt.ylabel('% of Missing Data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:34.410166Z","iopub.execute_input":"2022-11-13T16:10:34.411213Z","iopub.status.idle":"2022-11-13T16:10:42.484687Z","shell.execute_reply.started":"2022-11-13T16:10:34.411153Z","shell.execute_reply":"2022-11-13T16:10:42.483729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discard=nan[nan[\"Usability\"]!=\"Keep\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:42.515858Z","iopub.execute_input":"2022-11-13T16:10:42.516789Z","iopub.status.idle":"2022-11-13T16:10:42.526366Z","shell.execute_reply.started":"2022-11-13T16:10:42.516750Z","shell.execute_reply":"2022-11-13T16:10:42.525332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"delete_list=[]\nfor i in discard[\"feat\"]:\n    delete_list.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:43.070084Z","iopub.execute_input":"2022-11-13T16:10:43.071411Z","iopub.status.idle":"2022-11-13T16:10:43.077542Z","shell.execute_reply.started":"2022-11-13T16:10:43.071362Z","shell.execute_reply":"2022-11-13T16:10:43.076118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Feature Engineering</span>**","metadata":{}},{"cell_type":"code","source":"train=train_df.drop(delete_list,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:51.127556Z","iopub.execute_input":"2022-11-13T16:10:51.127913Z","iopub.status.idle":"2022-11-13T16:10:51.169107Z","shell.execute_reply.started":"2022-11-13T16:10:51.127881Z","shell.execute_reply":"2022-11-13T16:10:51.168135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:54.081803Z","iopub.execute_input":"2022-11-13T16:10:54.082866Z","iopub.status.idle":"2022-11-13T16:10:54.090239Z","shell.execute_reply.started":"2022-11-13T16:10:54.082818Z","shell.execute_reply":"2022-11-13T16:10:54.089159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=train_df.drop([\"customer_ID\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:10:57.505000Z","iopub.execute_input":"2022-11-13T16:10:57.505783Z","iopub.status.idle":"2022-11-13T16:10:57.558297Z","shell.execute_reply.started":"2022-11-13T16:10:57.505741Z","shell.execute_reply":"2022-11-13T16:10:57.557370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Splitting Train & Test</span>**","metadata":{}},{"cell_type":"code","source":"x=train.drop(\"target\",axis=1)\ny=train[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:11:00.018972Z","iopub.execute_input":"2022-11-13T16:11:00.019355Z","iopub.status.idle":"2022-11-13T16:11:00.045570Z","shell.execute_reply.started":"2022-11-13T16:11:00.019319Z","shell.execute_reply":"2022-11-13T16:11:00.044380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(x,y,test_size=0.2,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:11:00.734811Z","iopub.execute_input":"2022-11-13T16:11:00.735512Z","iopub.status.idle":"2022-11-13T16:11:00.858720Z","shell.execute_reply.started":"2022-11-13T16:11:00.735471Z","shell.execute_reply":"2022-11-13T16:11:00.857661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Null Value Imputation</span>**","metadata":{}},{"cell_type":"code","source":"si_num=SimpleImputer(strategy=\"median\")\nsi_cat=SimpleImputer(strategy=\"most_frequent\")","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:11:02.466418Z","iopub.execute_input":"2022-11-13T16:11:02.467054Z","iopub.status.idle":"2022-11-13T16:11:02.472576Z","shell.execute_reply.started":"2022-11-13T16:11:02.467014Z","shell.execute_reply":"2022-11-13T16:11:02.471455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = [col for col in x if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:11:03.408823Z","iopub.execute_input":"2022-11-13T16:11:03.409164Z","iopub.status.idle":"2022-11-13T16:11:03.414456Z","shell.execute_reply.started":"2022-11-13T16:11:03.409132Z","shell.execute_reply":"2022-11-13T16:11:03.413472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_tr=pd.DataFrame(si_num.fit_transform(x_train[num_df.columns]),columns=num_df.columns)\nx_te=pd.DataFrame(si_num.transform(x_test[num_df.columns]),columns=num_df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:14:49.327600Z","iopub.execute_input":"2022-11-13T16:14:49.327963Z","iopub.status.idle":"2022-11-13T16:14:51.308187Z","shell.execute_reply.started":"2022-11-13T16:14:49.327929Z","shell.execute_reply":"2022-11-13T16:14:51.307233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train_cat=pd.DataFrame(si_cat.fit_transform(x_train[cat_df.columns]),columns=cat_df.columns)\nx_test_cat=pd.DataFrame(si_cat.transform(x_test[cat_df.columns]),columns=cat_df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:15:32.787044Z","iopub.execute_input":"2022-11-13T16:15:32.787785Z","iopub.status.idle":"2022-11-13T16:15:32.821560Z","shell.execute_reply.started":"2022-11-13T16:15:32.787747Z","shell.execute_reply":"2022-11-13T16:15:32.820510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_train=pd.concat([x_tr,x_train_cat],axis=1)\nmain_test=pd.concat([x_te,x_test_cat],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:17:05.581549Z","iopub.execute_input":"2022-11-13T16:17:05.581958Z","iopub.status.idle":"2022-11-13T16:17:05.632594Z","shell.execute_reply.started":"2022-11-13T16:17:05.581923Z","shell.execute_reply":"2022-11-13T16:17:05.631547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:18:14.880022Z","iopub.execute_input":"2022-11-13T16:18:14.880483Z","iopub.status.idle":"2022-11-13T16:18:14.919202Z","shell.execute_reply.started":"2022-11-13T16:18:14.880440Z","shell.execute_reply":"2022-11-13T16:18:14.918250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Model Building</span>**","metadata":{}},{"cell_type":"code","source":"lr=DecisionTreeClassifier(max_depth=5)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:18:19.560103Z","iopub.execute_input":"2022-11-13T16:18:19.561036Z","iopub.status.idle":"2022-11-13T16:18:19.566717Z","shell.execute_reply.started":"2022-11-13T16:18:19.560987Z","shell.execute_reply":"2022-11-13T16:18:19.565385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr.fit(main_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:18:48.395431Z","iopub.execute_input":"2022-11-13T16:18:48.396096Z","iopub.status.idle":"2022-11-13T16:18:53.464677Z","shell.execute_reply.started":"2022-11-13T16:18:48.396037Z","shell.execute_reply":"2022-11-13T16:18:53.463504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=lr.predict(main_train)\npred1=lr.predict(main_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:19:20.901844Z","iopub.execute_input":"2022-11-13T16:19:20.902300Z","iopub.status.idle":"2022-11-13T16:19:21.013151Z","shell.execute_reply.started":"2022-11-13T16:19:20.902262Z","shell.execute_reply":"2022-11-13T16:19:21.012007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn. ensemble import RandomForestClassifier\n# create the random forest with your hyperparameters.\nmodel = RandomForestClassifier (n_estimators=100)\n# fit the model to start training.\n\nmodel. fit(main_train,y_train)\n# get the importance of the resulting features.\nimportances = model.feature_importances_\n# create a data frame for visualization.\nfinal_df = pd. DataFrame({\"Features\": pd.DataFrame(main_train).columns, \"Importances\": importances})\nfinal_df.set_index('Importances')\n\n# sort in ascending order to better visualization.\nfinal_df = final_df.sort_values(\"Importances\")\n# plot the feature importances in bars.\nfinal_df.plot.bar(color = 'teal' )\n\n","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:25:13.768126Z","iopub.execute_input":"2022-11-13T16:25:13.769069Z","iopub.status.idle":"2022-11-13T16:26:07.256979Z","shell.execute_reply.started":"2022-11-13T16:25:13.769031Z","shell.execute_reply":"2022-11-13T16:26:07.256068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=model.predict(main_train)\npred1=model.predict(main_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:29:17.898441Z","iopub.execute_input":"2022-11-13T16:29:17.898802Z","iopub.status.idle":"2022-11-13T16:29:18.388667Z","shell.execute_reply.started":"2022-11-13T16:29:17.898770Z","shell.execute_reply":"2022-11-13T16:29:18.387677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:33:43.581149Z","iopub.execute_input":"2022-11-13T16:33:43.581894Z","iopub.status.idle":"2022-11-13T16:33:43.589174Z","shell.execute_reply.started":"2022-11-13T16:33:43.581856Z","shell.execute_reply":"2022-11-13T16:33:43.588123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid = {#\"n_estimators\" : [50,100,150],\n              #\"min_samples_split\": [2, 5,7],\n              \"max_depth\": [3,4,5, 7],\n              #\"min_samples_leaf\": [1, 5],\n              #\"ccp_alpha\":[0.0001,0.001,0.01,0.1]\n             }","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:52:19.713883Z","iopub.execute_input":"2022-11-13T16:52:19.714609Z","iopub.status.idle":"2022-11-13T16:52:19.719288Z","shell.execute_reply.started":"2022-11-13T16:52:19.714571Z","shell.execute_reply":"2022-11-13T16:52:19.718178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rc=RandomForestClassifier()\nrfclf_grid = GridSearchCV(rc, param_grid, cv=3,verbose=1)\nrfclf_grid.fit(main_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:52:24.755416Z","iopub.execute_input":"2022-11-13T16:52:24.755771Z","iopub.status.idle":"2022-11-13T16:54:00.813198Z","shell.execute_reply.started":"2022-11-13T16:52:24.755740Z","shell.execute_reply":"2022-11-13T16:54:00.812245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=rfclf_grid.predict(main_train)\npred1=rfclf_grid.predict(main_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T17:03:32.886106Z","iopub.execute_input":"2022-11-13T17:03:32.887158Z","iopub.status.idle":"2022-11-13T17:03:33.177310Z","shell.execute_reply.started":"2022-11-13T17:03:32.887120Z","shell.execute_reply":"2022-11-13T17:03:33.176216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nparams = {'boosting_type': 'gbdt',\n              'n_estimators': 1000,\n              'num_leaves': 50,\n              'learning_rate': 0.05,\n              'colsample_bytree': 0.9,\n              'min_child_samples': 2000,\n              'max_bins': 500,\n              'reg_alpha': 2,\n              'objective': 'binary',\n              'random_state': 21}\ngbm = LGBMClassifier(**params).fit(main_train,y_train)\n                                       ","metadata":{"execution":{"iopub.status.busy":"2022-11-13T17:00:22.794830Z","iopub.execute_input":"2022-11-13T17:00:22.795218Z","iopub.status.idle":"2022-11-13T17:01:15.361730Z","shell.execute_reply.started":"2022-11-13T17:00:22.795165Z","shell.execute_reply":"2022-11-13T17:01:15.360921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred=gbm.predict(main_train)\npred1=gbm.predict(main_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T17:03:44.717471Z","iopub.execute_input":"2022-11-13T17:03:44.718157Z","iopub.status.idle":"2022-11-13T17:03:45.397021Z","shell.execute_reply.started":"2022-11-13T17:03:44.718123Z","shell.execute_reply":"2022-11-13T17:03:45.395707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:#87CEEB;\">Model Evaluation</span>**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true, y_pred, return_components=False) -> float:\n    \"\"\"Amex metric for ndarrays\"\"\"\n    def top_four_percent_captured(df) -> float:\n        \"\"\"Corresponds to the recall for a threshold of 4 %\"\"\"\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(df) -> float:\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(df) -> float:\n        \"\"\"Corresponds to 2 * AUC - 1\"\"\"\n        df2 = pd.DataFrame({'target': df.target, 'prediction': df.target})\n        df2.sort_values('prediction', ascending=False, inplace=True)\n        return weighted_gini(df) / weighted_gini(df2)\n\n    df = pd.DataFrame({'target': y_true.ravel(), 'prediction': y_pred.ravel()})\n    df.sort_values('prediction', ascending=False, inplace=True)\n    g = normalized_weighted_gini(df)\n    d = top_four_percent_captured(df)\n\n    if return_components: return g, d, 0.5 * (g + d)\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T16:20:18.643961Z","iopub.execute_input":"2022-11-13T16:20:18.644397Z","iopub.status.idle":"2022-11-13T16:20:18.657224Z","shell.execute_reply.started":"2022-11-13T16:20:18.644358Z","shell.execute_reply":"2022-11-13T16:20:18.655977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_train,pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T17:03:48.570639Z","iopub.execute_input":"2022-11-13T17:03:48.571042Z","iopub.status.idle":"2022-11-13T17:03:48.607266Z","shell.execute_reply.started":"2022-11-13T17:03:48.571009Z","shell.execute_reply":"2022-11-13T17:03:48.605627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_test,pred1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T17:03:49.531735Z","iopub.execute_input":"2022-11-13T17:03:49.532886Z","iopub.status.idle":"2022-11-13T17:03:49.560166Z","shell.execute_reply.started":"2022-11-13T17:03:49.532837Z","shell.execute_reply":"2022-11-13T17:03:49.558995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}