{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns# data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-21T19:12:16.788732Z","iopub.execute_input":"2023-02-21T19:12:16.789422Z","iopub.status.idle":"2023-02-21T19:12:17.861130Z","shell.execute_reply.started":"2023-02-21T19:12:16.789336Z","shell.execute_reply":"2023-02-21T19:12:17.860024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_data = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=1000000)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:12:17.863030Z","iopub.execute_input":"2023-02-21T19:12:17.863384Z","iopub.status.idle":"2023-02-21T19:13:38.647129Z","shell.execute_reply.started":"2023-02-21T19:12:17.863354Z","shell.execute_reply":"2023-02-21T19:13:38.645761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', nrows = 1000000)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:13:38.648793Z","iopub.execute_input":"2023-02-21T19:13:38.649204Z","iopub.status.idle":"2023-02-21T19:13:39.653244Z","shell.execute_reply.started":"2023-02-21T19:13:38.649173Z","shell.execute_reply":"2023-02-21T19:13:39.652081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.merge(df_train_data, df_train_labels, how=\"inner\", on=[\"customer_ID\"])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:13:39.655571Z","iopub.execute_input":"2023-02-21T19:13:39.655943Z","iopub.status.idle":"2023-02-21T19:13:43.506909Z","shell.execute_reply.started":"2023-02-21T19:13:39.655910Z","shell.execute_reply":"2023-02-21T19:13:43.505536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:13:43.508338Z","iopub.execute_input":"2023-02-21T19:13:43.508659Z","iopub.status.idle":"2023-02-21T19:13:43.516176Z","shell.execute_reply.started":"2023-02-21T19:13:43.508629Z","shell.execute_reply":"2023-02-21T19:13:43.514670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:13:43.517953Z","iopub.execute_input":"2023-02-21T19:13:43.518422Z","iopub.status.idle":"2023-02-21T19:13:43.531820Z","shell.execute_reply.started":"2023-02-21T19:13:43.518382Z","shell.execute_reply":"2023-02-21T19:13:43.530720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:13:43.533988Z","iopub.execute_input":"2023-02-21T19:13:43.534916Z","iopub.status.idle":"2023-02-21T19:14:08.685851Z","shell.execute_reply.started":"2023-02-21T19:13:43.534873Z","shell.execute_reply":"2023-02-21T19:14:08.684202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"features_with_na = [features for features in df_train.columns if df_train[features].isnull().sum() > 1]\n\n# Calculate missing value rate for each feature\nfor feature in features_with_na:\n    if feature=='D_64':\n        pass\n    else:\n        mdata = df_train[feature].astype(float)\n        missing_value_round = np.round(mdata.isnull().mean(),4)\n        print(feature, missing_value_round)  \nprint(len(features_with_na))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:08.687805Z","iopub.execute_input":"2023-02-21T19:14:08.688314Z","iopub.status.idle":"2023-02-21T19:14:09.670708Z","shell.execute_reply.started":"2023-02-21T19:14:08.688282Z","shell.execute_reply":"2023-02-21T19:14:09.669476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in df_train.columns:\n    # Find the number of cells in the column that contain the string 'missing value'\n    print(col, df_train[col].isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:09.672205Z","iopub.execute_input":"2023-02-21T19:14:09.674507Z","iopub.status.idle":"2023-02-21T19:14:10.232080Z","shell.execute_reply.started":"2023-02-21T19:14:09.674472Z","shell.execute_reply":"2023-02-21T19:14:10.230940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_cols = [i for i in df_train.columns if df_train[i].isnull().any()]\nprint(nan_cols)\nprint(len(nan_cols))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:10.236317Z","iopub.execute_input":"2023-02-21T19:14:10.236627Z","iopub.status.idle":"2023-02-21T19:14:10.581299Z","shell.execute_reply.started":"2023-02-21T19:14:10.236599Z","shell.execute_reply":"2023-02-21T19:14:10.580062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:10.582999Z","iopub.execute_input":"2023-02-21T19:14:10.583305Z","iopub.status.idle":"2023-02-21T19:14:10.603226Z","shell.execute_reply.started":"2023-02-21T19:14:10.583279Z","shell.execute_reply":"2023-02-21T19:14:10.602137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:10.604801Z","iopub.execute_input":"2023-02-21T19:14:10.605276Z","iopub.status.idle":"2023-02-21T19:14:20.582933Z","shell.execute_reply.started":"2023-02-21T19:14:10.605247Z","shell.execute_reply":"2023-02-21T19:14:20.582119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:20.584202Z","iopub.execute_input":"2023-02-21T19:14:20.585090Z","iopub.status.idle":"2023-02-21T19:14:30.829998Z","shell.execute_reply.started":"2023-02-21T19:14:20.585057Z","shell.execute_reply":"2023-02-21T19:14:30.829026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train.columns:\n    print(i)\n    print(df_train[i].unique())\n    print(df_train[i].value_counts())\n    print('\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:14:30.831315Z","iopub.execute_input":"2023-02-21T19:14:30.831922Z","iopub.status.idle":"2023-02-21T19:15:15.949460Z","shell.execute_reply.started":"2023-02-21T19:14:30.831884Z","shell.execute_reply":"2023-02-21T19:15:15.948471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features=[feature for feature in df_train.columns if df_train[feature].dtypes !='O']\nprint(\"Number of numerical values: \",len(numerical_features))\ndf_train[numerical_features].head()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:15:15.950769Z","iopub.execute_input":"2023-02-21T19:15:15.951410Z","iopub.status.idle":"2023-02-21T19:15:16.431983Z","shell.execute_reply.started":"2023-02-21T19:15:15.951378Z","shell.execute_reply":"2023-02-21T19:15:16.430831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"variable_features=[feature for feature in df_train.columns if df_train[feature].dtypes =='O']\nprint(\"Number of numerical values: \",len(variable_features))\ndf_train[variable_features].head()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:15:16.433605Z","iopub.execute_input":"2023-02-21T19:15:16.433999Z","iopub.status.idle":"2023-02-21T19:15:16.481166Z","shell.execute_reply.started":"2023-02-21T19:15:16.433968Z","shell.execute_reply":"2023-02-21T19:15:16.480089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train['D_63'])\nprint(df_train['D_63'].unique())\nprint(df_train['D_63'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:15:16.482561Z","iopub.execute_input":"2023-02-21T19:15:16.484018Z","iopub.status.idle":"2023-02-21T19:15:16.613201Z","shell.execute_reply.started":"2023-02-21T19:15:16.483977Z","shell.execute_reply":"2023-02-21T19:15:16.612418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train['D_64'])\nprint(df_train['D_64'].unique())\nprint(df_train['D_64'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:15:16.614649Z","iopub.execute_input":"2023-02-21T19:15:16.615008Z","iopub.status.idle":"2023-02-21T19:15:16.703522Z","shell.execute_reply.started":"2023-02-21T19:15:16.614977Z","shell.execute_reply":"2023-02-21T19:15:16.702409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train['S_2'])\nprint(df_train['S_2'].unique())\nprint(df_train['S_2'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:15:16.704976Z","iopub.execute_input":"2023-02-21T19:15:16.705430Z","iopub.status.idle":"2023-02-21T19:15:16.853869Z","shell.execute_reply.started":"2023-02-21T19:15:16.705388Z","shell.execute_reply":"2023-02-21T19:15:16.852709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#how features with no value affect the target\nncols = 3\nnrows = (len(features_with_na) + ncols - 1) // ncols\nfig, axs = plt.subplots(nrows, ncols, figsize=(15, 85))\naxs = axs.flatten()\nfor i, feature in enumerate(features_with_na):\n    # Copy the data\n    data1 = df_train.copy()\n    data1[feature] = np.where(data1[feature].isnull(), 1, 0)\n    data1.groupby(feature)['target'].mean().plot.bar(ax=axs[i])\n    axs[i].set_title(feature)\n\nfor i in range(len(features_with_na), len(axs)):\n    fig.delaxes(axs[i])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:15:16.855155Z","iopub.execute_input":"2023-02-21T19:15:16.855563Z","iopub.status.idle":"2023-02-21T19:17:33.890608Z","shell.execute_reply.started":"2023-02-21T19:15:16.855534Z","shell.execute_reply":"2023-02-21T19:17:33.889653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aggregated_data = df_train.groupby('S_2')['target'].mean()\n\naggregated_data.plot(kind='line', figsize=(15, 10))\n\nplt.xlabel('S_2')\nplt.ylabel('Defaulter or paid cerdit money')\nplt.title(\"American-Express-Default\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:33.891928Z","iopub.execute_input":"2023-02-21T19:17:33.892882Z","iopub.status.idle":"2023-02-21T19:17:34.273099Z","shell.execute_reply.started":"2023-02-21T19:17:33.892847Z","shell.execute_reply":"2023-02-21T19:17:34.271947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:34.274819Z","iopub.execute_input":"2023-02-21T19:17:34.275277Z","iopub.status.idle":"2023-02-21T19:17:34.281381Z","shell.execute_reply.started":"2023-02-21T19:17:34.275234Z","shell.execute_reply":"2023-02-21T19:17:34.280296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_cat = df_train[['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:34.283105Z","iopub.execute_input":"2023-02-21T19:17:34.283548Z","iopub.status.idle":"2023-02-21T19:17:34.324039Z","shell.execute_reply.started":"2023-02-21T19:17:34.283497Z","shell.execute_reply":"2023-02-21T19:17:34.323028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train_cat.columns:\n    print(i)\n    print(df_train_cat[i].unique())\n    print(df_train_cat[i].value_counts())\n    print('\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:34.325325Z","iopub.execute_input":"2023-02-21T19:17:34.325668Z","iopub.status.idle":"2023-02-21T19:17:34.722202Z","shell.execute_reply.started":"2023-02-21T19:17:34.325638Z","shell.execute_reply":"2023-02-21T19:17:34.721374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:34.723440Z","iopub.execute_input":"2023-02-21T19:17:34.724452Z","iopub.status.idle":"2023-02-21T19:17:34.730149Z","shell.execute_reply.started":"2023-02-21T19:17:34.724418Z","shell.execute_reply":"2023-02-21T19:17:34.729206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train_cat.columns:\n    plt.figure(figsize=(15,6))\n    sns.countplot(df_train_cat[i], data = df_train_cat, palette = 'hls')\n    plt.xticks(rotation = 90)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:34.731798Z","iopub.execute_input":"2023-02-21T19:17:34.732325Z","iopub.status.idle":"2023-02-21T19:17:44.236714Z","shell.execute_reply.started":"2023-02-21T19:17:34.732284Z","shell.execute_reply":"2023-02-21T19:17:44.235657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train_cat.columns:\n    plt.figure(figsize=(20,8))\n    df_train_cat[i].value_counts().plot(kind = 'pie',autopct='%1.1f%%', startangle=90)\n    plt.xticks(rotation = 90)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:44.237987Z","iopub.execute_input":"2023-02-21T19:17:44.238325Z","iopub.status.idle":"2023-02-21T19:17:46.526312Z","shell.execute_reply.started":"2023-02-21T19:17:44.238294Z","shell.execute_reply":"2023-02-21T19:17:46.524782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(df_train.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:46.545361Z","iopub.execute_input":"2023-02-21T19:17:46.546390Z","iopub.status.idle":"2023-02-21T19:17:47.106664Z","shell.execute_reply.started":"2023-02-21T19:17:46.546329Z","shell.execute_reply":"2023-02-21T19:17:47.105535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_colors = [\"#ffd670\",\"#70d6ff\",\"#ff4d6d\",\"#8338ec\",\"#90cf8e\"]\ncustomPalette = sns.set_palette(sns.color_palette(custom_colors))\nsns.palplot(sns.color_palette(custom_colors),size=1.2)\nplt.tick_params(axis='both', labelsize=0, length = 0)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:47.108411Z","iopub.execute_input":"2023-02-21T19:17:47.108773Z","iopub.status.idle":"2023-02-21T19:17:47.190909Z","shell.execute_reply.started":"2023-02-21T19:17:47.108741Z","shell.execute_reply":"2023-02-21T19:17:47.189119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"background_color = 'white'\nmissing = pd.DataFrame(columns = ['% Missing values'],data = df_train.isnull().sum()/len(df_train))\nfig = plt.figure(figsize = (20, 60),facecolor=background_color)\ngs = fig.add_gridspec(1, 2)\ngs.update(wspace = 0.5, hspace = 0.5)\nax0 = fig.add_subplot(gs[0, 0])\nfor s in [\"right\", \"top\",\"bottom\",\"left\"]:\n    ax0.spines[s].set_visible(False)\nsns.heatmap(missing,cbar = False,annot = True,fmt =\".2%\", linewidths = 2,cmap = custom_colors,vmax = 1, ax = ax0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:47.193560Z","iopub.execute_input":"2023-02-21T19:17:47.194190Z","iopub.status.idle":"2023-02-21T19:17:50.952162Z","shell.execute_reply.started":"2023-02-21T19:17:47.194135Z","shell.execute_reply":"2023-02-21T19:17:50.951023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:50.953807Z","iopub.execute_input":"2023-02-21T19:17:50.954155Z","iopub.status.idle":"2023-02-21T19:17:51.437447Z","shell.execute_reply.started":"2023-02-21T19:17:50.954125Z","shell.execute_reply":"2023-02-21T19:17:51.436475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_Delinquency = [c for c in df_train.columns if c.startswith('D_')]\nfeat_Spend = [c for c in df_train.columns if c.startswith('S_')]\nfeat_Payment = [c for c in df_train.columns if c.startswith('P_')]\nfeat_Balance = [c for c in df_train.columns if c.startswith('B_')]\nfeat_Risk = [c for c in df_train.columns if c.startswith('R_')]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:51.438921Z","iopub.execute_input":"2023-02-21T19:17:51.439356Z","iopub.status.idle":"2023-02-21T19:17:51.446766Z","shell.execute_reply.started":"2023-02-21T19:17:51.439304Z","shell.execute_reply":"2023-02-21T19:17:51.445412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Total number of Delinquency variables: {len(feat_Delinquency)}')\nprint(f'Total number of Spend variables: {len(feat_Spend)}')\nprint(f'Total number of Payment variables: {len(feat_Payment)}')\nprint(f'Total number of Balance variables: {len(feat_Balance)}')\nprint(f'Total number of Risk variables: {len(feat_Risk)}')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:51.448464Z","iopub.execute_input":"2023-02-21T19:17:51.448804Z","iopub.status.idle":"2023-02-21T19:17:51.461744Z","shell.execute_reply.started":"2023-02-21T19:17:51.448774Z","shell.execute_reply":"2023-02-21T19:17:51.460540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=['Delinquency', 'Spend','Payment','Balance','Risk']\nvalues= [len(feat_Delinquency), len(feat_Spend),len(feat_Payment), len(feat_Balance),len(feat_Risk)]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:51.463324Z","iopub.execute_input":"2023-02-21T19:17:51.463663Z","iopub.status.idle":"2023-02-21T19:17:51.479958Z","shell.execute_reply.started":"2023-02-21T19:17:51.463633Z","shell.execute_reply":"2023-02-21T19:17:51.478772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nimport plotly.express as px\nfrom itertools import cycle","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:51.481203Z","iopub.execute_input":"2023-02-21T19:17:51.482050Z","iopub.status.idle":"2023-02-21T19:17:52.608248Z","shell.execute_reply.started":"2023-02-21T19:17:51.482017Z","shell.execute_reply":"2023-02-21T19:17:52.607037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig_1 = go.Figure()\nfig_1.add_trace(go.Pie(values = values,labels = labels,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig_1.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#70d6ff\",\"#ff9770\"]),\n                  title = dict(text = 'Feature Distribution'))  \nfig_1.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:52.610955Z","iopub.execute_input":"2023-02-21T19:17:52.611672Z","iopub.status.idle":"2023-02-21T19:17:52.719427Z","shell.execute_reply.started":"2023-02-21T19:17:52.611627Z","shell.execute_reply":"2023-02-21T19:17:52.718317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:52.722814Z","iopub.execute_input":"2023-02-21T19:17:52.723321Z","iopub.status.idle":"2023-02-21T19:17:52.732385Z","shell.execute_reply.started":"2023-02-21T19:17:52.723289Z","shell.execute_reply":"2023-02-21T19:17:52.731086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nsns.countplot(df_train['target'], data = df_train, palette = 'hls')\nplt.xticks(rotation = 90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:52.733835Z","iopub.execute_input":"2023-02-21T19:17:52.734333Z","iopub.status.idle":"2023-02-21T19:17:53.050360Z","shell.execute_reply.started":"2023-02-21T19:17:52.734304Z","shell.execute_reply":"2023-02-21T19:17:53.049250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,8))\ndf_train['target'].value_counts().plot(kind = 'pie',autopct='%1.1f%%', startangle=90)\nplt.xticks(rotation = 90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.051664Z","iopub.execute_input":"2023-02-21T19:17:53.052254Z","iopub.status.idle":"2023-02-21T19:17:53.180526Z","shell.execute_reply.started":"2023-02-21T19:17:53.052190Z","shell.execute_reply":"2023-02-21T19:17:53.178993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_class = pd.DataFrame({'count': df_train.target.value_counts(),\n                             'percentage': df_train['target'].value_counts() / df_train.shape[0] * 100\n})","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.183011Z","iopub.execute_input":"2023-02-21T19:17:53.183620Z","iopub.status.idle":"2023-02-21T19:17:53.197907Z","shell.execute_reply.started":"2023-02-21T19:17:53.183557Z","shell.execute_reply":"2023-02-21T19:17:53.196473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_class ","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.200317Z","iopub.execute_input":"2023-02-21T19:17:53.200917Z","iopub.status.idle":"2023-02-21T19:17:53.219180Z","shell.execute_reply.started":"2023-02-21T19:17:53.200860Z","shell.execute_reply":"2023-02-21T19:17:53.217759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace(go.Pie(values = target_class['count'],labels = target_class.index,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#90cf8e\",\"#ff70a6\"]),\n                  title = dict(text = 'Target Distribution'))  \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.221751Z","iopub.execute_input":"2023-02-21T19:17:53.222633Z","iopub.status.idle":"2023-02-21T19:17:53.245444Z","shell.execute_reply.started":"2023-02-21T19:17:53.222577Z","shell.execute_reply":"2023-02-21T19:17:53.244598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stat_plot = df_train.reset_index().groupby('S_2')['customer_ID'].nunique().reset_index()\nfig = go.Figure()\nfig.add_trace(go.Scatter(x = stat_plot['S_2'], y = stat_plot['customer_ID']))\nfig.update_layout(title=\"Customer Statements\", width = 800, height = 600,xaxis_title ='Statement Date',\n                  paper_bgcolor='rgb(0,0,0,0)',plot_bgcolor='rgb(0,0,0,0)') \nfig['data'][0]['line']['color']=\"#ff9770\"\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.246618Z","iopub.execute_input":"2023-02-21T19:17:53.247105Z","iopub.status.idle":"2023-02-21T19:17:53.443208Z","shell.execute_reply.started":"2023-02-21T19:17:53.247076Z","shell.execute_reply":"2023-02-21T19:17:53.442128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop('S_2', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.445356Z","iopub.execute_input":"2023-02-21T19:17:53.445718Z","iopub.status.idle":"2023-02-21T19:17:53.545425Z","shell.execute_reply.started":"2023-02-21T19:17:53.445669Z","shell.execute_reply":"2023-02-21T19:17:53.544221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_cols = [c for c in df_train.columns if (c.startswith(('D','t'))) & (c not in cat_cols)]\ndf_del = df_train[del_cols]\nspd_cols = [c for c in df_train.columns if (c.startswith(('S','t'))) & (c not in cat_cols)]\ndf_spd = df_train[spd_cols]\npay_cols = [c for c in df_train.columns if (c.startswith(('P','t'))) & (c not in cat_cols)]\ndf_pay = df_train[pay_cols]\nbal_cols = [c for c in df_train.columns if (c.startswith(('B','t'))) & (c not in cat_cols)]\ndf_bal = df_train[bal_cols]\nris_cols = [c for c in df_train.columns if (c.startswith(('R','t'))) & (c not in cat_cols)]\ndf_ris = df_train[ris_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.546793Z","iopub.execute_input":"2023-02-21T19:17:53.547328Z","iopub.status.idle":"2023-02-21T19:17:53.589904Z","shell.execute_reply.started":"2023-02-21T19:17:53.547279Z","shell.execute_reply":"2023-02-21T19:17:53.588653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(29, 3, figsize = (35,150))\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(del_cols) - 1:\n        sns.kdeplot(x = del_cols[i], hue='target', data = df_del, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nfig.suptitle('Distribution of Delinquency Variables', fontsize = 35, x = 0.5, y = 1)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:17:53.591417Z","iopub.execute_input":"2023-02-21T19:17:53.591842Z","iopub.status.idle":"2023-02-21T19:18:45.372519Z","shell.execute_reply.started":"2023-02-21T19:17:53.591810Z","shell.execute_reply":"2023-02-21T19:18:45.371057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize =(11,11))\ncorr = df_del.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths =.6, cmap = custom_colors)\nplt.title('Correlation of Delinquency Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:18:45.374260Z","iopub.execute_input":"2023-02-21T19:18:45.374649Z","iopub.status.idle":"2023-02-21T19:18:47.846410Z","shell.execute_reply.started":"2023-02-21T19:18:45.374616Z","shell.execute_reply":"2023-02-21T19:18:47.845225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(8, 3, figsize = (16,18))\nfig.suptitle('Distribution of Spend Variables', fontsize = 15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(spd_cols) - 1:\n        sns.kdeplot(x = spd_cols[i], hue ='target', data = df_spd, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:18:47.848259Z","iopub.execute_input":"2023-02-21T19:18:47.848633Z","iopub.status.idle":"2023-02-21T19:19:02.062596Z","shell.execute_reply.started":"2023-02-21T19:18:47.848602Z","shell.execute_reply":"2023-02-21T19:19:02.061436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"S_cols = [c for c in df_train.columns if (c.startswith(('S')))]\ndf_S = df_train[S_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:02.063850Z","iopub.execute_input":"2023-02-21T19:19:02.064171Z","iopub.status.idle":"2023-02-21T19:19:02.073738Z","shell.execute_reply.started":"2023-02-21T19:19:02.064142Z","shell.execute_reply":"2023-02-21T19:19:02.072658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\ncorr = df_S.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths = .6, cmap = custom_colors)\nplt.title('Correlation of Spend Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:02.075685Z","iopub.execute_input":"2023-02-21T19:19:02.076112Z","iopub.status.idle":"2023-02-21T19:19:02.753975Z","shell.execute_reply.started":"2023-02-21T19:19:02.076081Z","shell.execute_reply":"2023-02-21T19:19:02.752603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize = (12,4))\nfig.suptitle('Distribution of Payment Variables',fontsize = 15)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(pay_cols) - 1:\n        sns.kdeplot(x = pay_cols[i], hue ='target', data = df_pay, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:02.755868Z","iopub.execute_input":"2023-02-21T19:19:02.756433Z","iopub.status.idle":"2023-02-21T19:19:04.951375Z","shell.execute_reply.started":"2023-02-21T19:19:02.756400Z","shell.execute_reply":"2023-02-21T19:19:04.950466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"P_cols = [c for c in df_train.columns if (c.startswith(('P')))]\ndf_P = df_train[P_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:04.952824Z","iopub.execute_input":"2023-02-21T19:19:04.953737Z","iopub.status.idle":"2023-02-21T19:19:04.960149Z","shell.execute_reply.started":"2023-02-21T19:19:04.953658Z","shell.execute_reply":"2023-02-21T19:19:04.959036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (6,6))\ncorr = df_P.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths = .6, cmap = custom_colors)\nplt.title('Correlation of Payment Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:04.961720Z","iopub.execute_input":"2023-02-21T19:19:04.962111Z","iopub.status.idle":"2023-02-21T19:19:05.209568Z","shell.execute_reply.started":"2023-02-21T19:19:04.962082Z","shell.execute_reply":"2023-02-21T19:19:05.208681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 4, figsize = (15,24))\nfig.suptitle('Distribution of Balance Variables',fontsize = 15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(bal_cols) - 1:\n        sns.kdeplot(x = bal_cols[i], hue ='target', data = df_bal, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:05.210843Z","iopub.execute_input":"2023-02-21T19:19:05.211334Z","iopub.status.idle":"2023-02-21T19:19:30.100529Z","shell.execute_reply.started":"2023-02-21T19:19:05.211304Z","shell.execute_reply":"2023-02-21T19:19:30.099254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"B_cols = [c for c in df_train.columns if (c.startswith(('B')))]\ndf_B = df_train[B_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:30.102161Z","iopub.execute_input":"2023-02-21T19:19:30.102524Z","iopub.status.idle":"2023-02-21T19:19:30.115600Z","shell.execute_reply.started":"2023-02-21T19:19:30.102491Z","shell.execute_reply":"2023-02-21T19:19:30.114373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\ncorr = df_B.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust=True, center = 0,square = True, linewidths =.6, cmap = custom_colors)\nplt.title('Correlation of Balance Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:30.117443Z","iopub.execute_input":"2023-02-21T19:19:30.117831Z","iopub.status.idle":"2023-02-21T19:19:31.351768Z","shell.execute_reply.started":"2023-02-21T19:19:30.117798Z","shell.execute_reply":"2023-02-21T19:19:31.350286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 3, figsize = (18,23))\nfig.suptitle('Distribution of Risk Variables',fontsize=15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(ris_cols) - 1:\n        sns.kdeplot(x = ris_cols[i], hue ='target', data = df_ris, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:31.353352Z","iopub.execute_input":"2023-02-21T19:19:31.353766Z","iopub.status.idle":"2023-02-21T19:19:51.694833Z","shell.execute_reply.started":"2023-02-21T19:19:31.353731Z","shell.execute_reply":"2023-02-21T19:19:51.693303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"R_cols = [c for c in df_train.columns if (c.startswith(('R')))]\ndf_R = df_train[R_cols]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:51.696849Z","iopub.execute_input":"2023-02-21T19:19:51.697352Z","iopub.status.idle":"2023-02-21T19:19:51.712047Z","shell.execute_reply.started":"2023-02-21T19:19:51.697306Z","shell.execute_reply":"2023-02-21T19:19:51.709852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(11,11))\ncorr = df_R.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0, square = True, linewidths =.6, cmap = custom_colors)\nplt.title('Correlation of Risk Variables')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:51.714152Z","iopub.execute_input":"2023-02-21T19:19:51.714631Z","iopub.status.idle":"2023-02-21T19:19:52.647658Z","shell.execute_reply.started":"2023-02-21T19:19:51.714582Z","shell.execute_reply":"2023-02-21T19:19:52.646408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"palette = cycle([\"#ffd670\",\"#70d6ff\",\"#ff4d6d\",\"#8338ec\",\"#90cf8e\"])\ntarg = df_train.corrwith(df_train['target'], axis=0)\nval = [str(round(v ,1) *100) + '%' for v in targ.values]\nfig = go.Figure()\nfig.add_trace(go.Bar(y=targ.index, x= targ.values, orientation='h',text = val, marker_color = next(palette)))\nfig.update_layout(title = \"Correlation of variables with Target\",width = 750, height = 3500,\n                  paper_bgcolor='rgb(0,0,0,0)',plot_bgcolor='rgb(0,0,0,0)')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:52.649207Z","iopub.execute_input":"2023-02-21T19:19:52.649617Z","iopub.status.idle":"2023-02-21T19:19:52.989456Z","shell.execute_reply.started":"2023-02-21T19:19:52.649583Z","shell.execute_reply":"2023-02-21T19:19:52.988312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Coding with PCA","metadata":{}},{"cell_type":"code","source":"amex_data = pd.merge(df_train_data, df_train_labels, how=\"inner\", on=[\"customer_ID\"])\namex_labels = amex_data[['customer_ID', 'target']]\namex_labels = amex_labels.groupby('customer_ID').agg({'target' : 'min'})","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:52.990751Z","iopub.execute_input":"2023-02-21T19:19:52.991098Z","iopub.status.idle":"2023-02-21T19:19:58.277187Z","shell.execute_reply.started":"2023-02-21T19:19:52.991070Z","shell.execute_reply":"2023-02-21T19:19:58.275767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_data = amex_data.drop(cat_cols , axis=1)\namex_data = amex_data.drop(['target'], axis=1)\nprint(\"The train data shape without the categorical features is:\", amex_data.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:58.278962Z","iopub.execute_input":"2023-02-21T19:19:58.279513Z","iopub.status.idle":"2023-02-21T19:19:59.243801Z","shell.execute_reply.started":"2023-02-21T19:19:58.279467Z","shell.execute_reply":"2023-02-21T19:19:59.242481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing data with high missing data\nmissing_data=pd.DataFrame((amex_data.isnull().sum()/len(amex_data))*100, columns=['% Missing'])\ncols_with_high_missing_data = list(missing_data[missing_data[\"% Missing\"] >80].index)\namex_data = amex_data.drop(cols_with_high_missing_data, axis=1)\nprint(\"Follwing columns have been removed from the dataset:\", cols_with_high_missing_data)\nprint(\"The amex data shape after removing features with high missing data is:\", amex_data.shape)\n\n#removing non numericals data\nprint('Non-numeric variables:', set(amex_data.columns) - set(amex_data._get_numeric_data().columns))\namex_data = amex_data.drop(['S_2'], axis=1)\nprint(\"The amex data shape after eliminating date variables:\", amex_data.shape)\n\n#data aggregation \naggregation_dict = dict.fromkeys(list(set(amex_data.columns) - set(['customer_ID'])), \"mean\")\nagg_amex_data = amex_data.groupby(['customer_ID']).agg(aggregation_dict)\nagg_amex_data.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:19:59.245500Z","iopub.execute_input":"2023-02-21T19:19:59.245896Z","iopub.status.idle":"2023-02-21T19:20:03.442657Z","shell.execute_reply.started":"2023-02-21T19:19:59.245861Z","shell.execute_reply":"2023-02-21T19:20:03.440971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing value treatment - Filling missing data with corresponding column Median\nfor column in agg_amex_data.columns:\n    median = agg_amex_data[column].median()\n    agg_amex_data[column] = agg_amex_data[column].fillna(median)\n    \n    \nagg_amex_data.head(5)  ","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:20:03.445047Z","iopub.execute_input":"2023-02-21T19:20:03.445630Z","iopub.status.idle":"2023-02-21T19:20:03.930457Z","shell.execute_reply.started":"2023-02-21T19:20:03.445575Z","shell.execute_reply":"2023-02-21T19:20:03.928532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Reduction","metadata":{}},{"cell_type":"code","source":"## Grouping columns by categories\n\nDeliquency_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='D']\nSpend_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='S']\nPayment_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='P']\nBalance_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='B']\nRisk_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='R']","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:20:03.932352Z","iopub.execute_input":"2023-02-21T19:20:03.932827Z","iopub.status.idle":"2023-02-21T19:20:03.942564Z","shell.execute_reply.started":"2023-02-21T19:20:03.932786Z","shell.execute_reply":"2023-02-21T19:20:03.940804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import scale\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.decomposition import IncrementalPCA","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:20:03.944537Z","iopub.execute_input":"2023-02-21T19:20:03.945170Z","iopub.status.idle":"2023-02-21T19:20:04.254743Z","shell.execute_reply.started":"2023-02-21T19:20:03.945130Z","shell.execute_reply":"2023-02-21T19:20:04.253555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## PCA on Deliqunecy\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Deliquency_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \nx=pd.DataFrame(StandardScaler().fit_transform(x))\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=70)\npcs.fit(x)\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(range(1, 71), [0.9] * 70, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 71)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:51:39.709524Z","iopub.execute_input":"2023-02-21T19:51:39.710785Z","iopub.status.idle":"2023-02-21T19:51:42.957779Z","shell.execute_reply.started":"2023-02-21T19:51:39.710728Z","shell.execute_reply":"2023-02-21T19:51:42.956284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Deliquency_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\ndeliquency_variables = pca_features[['Deliquency_{}'.format(i) for i in range(1, 39 )]]\ndeliquency_variables","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:52:13.375528Z","iopub.execute_input":"2023-02-21T19:52:13.375962Z","iopub.status.idle":"2023-02-21T19:52:13.909540Z","shell.execute_reply.started":"2023-02-21T19:52:13.375929Z","shell.execute_reply":"2023-02-21T19:52:13.908367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## PCA on Spend\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Spend_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \nx=pd.DataFrame(StandardScaler().fit_transform(x))\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=20)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,20 )), [0.9] * 20, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 21)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:53:19.779074Z","iopub.execute_input":"2023-02-21T19:53:19.779831Z","iopub.status.idle":"2023-02-21T19:53:21.448910Z","shell.execute_reply.started":"2023-02-21T19:53:19.779792Z","shell.execute_reply":"2023-02-21T19:53:21.447738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Spend_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\nspend_variables = pca_features[['Spend_{}'.format(i) for i in range(1, 9 )]]\nspend_variables","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:20:08.675201Z","iopub.execute_input":"2023-02-21T19:20:08.675768Z","iopub.status.idle":"2023-02-21T19:20:08.796718Z","shell.execute_reply.started":"2023-02-21T19:20:08.675714Z","shell.execute_reply":"2023-02-21T19:20:08.794784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## PCA on Payment\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Payment_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \nx=pd.DataFrame(StandardScaler().fit_transform(x))\n\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=3)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,3 )), [0.9] * 3, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 4)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:53:44.650069Z","iopub.execute_input":"2023-02-21T19:53:44.650482Z","iopub.status.idle":"2023-02-21T19:53:45.178371Z","shell.execute_reply.started":"2023-02-21T19:53:44.650451Z","shell.execute_reply":"2023-02-21T19:53:45.177150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Payment_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\npayment_variables = pca_features[['Payment_{}'.format(i) for i in range(1, 4 )]]\npayment_variables","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:53:52.951465Z","iopub.execute_input":"2023-02-21T19:53:52.951912Z","iopub.status.idle":"2023-02-21T19:53:52.996644Z","shell.execute_reply.started":"2023-02-21T19:53:52.951876Z","shell.execute_reply":"2023-02-21T19:53:52.994974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## PCA on Balance\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Balance_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \nx=pd.DataFrame(StandardScaler().fit_transform(x))\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=20)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,20 )), [0.9] * 20, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:54:06.147105Z","iopub.execute_input":"2023-02-21T19:54:06.147526Z","iopub.status.idle":"2023-02-21T19:54:08.142794Z","shell.execute_reply.started":"2023-02-21T19:54:06.147493Z","shell.execute_reply":"2023-02-21T19:54:08.141774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Balance_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\nbalance_variables = pca_features[['Balance_{}'.format(i) for i in range(1, 4 )]]\nbalance_variables","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:54:23.083159Z","iopub.execute_input":"2023-02-21T19:54:23.084141Z","iopub.status.idle":"2023-02-21T19:54:23.893124Z","shell.execute_reply.started":"2023-02-21T19:54:23.084103Z","shell.execute_reply":"2023-02-21T19:54:23.892101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## PCA on Risk\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Risk_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \nx=pd.DataFrame(StandardScaler().fit_transform(x))\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=20)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,20 )), [0.9] * 20, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 21)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:54:40.214869Z","iopub.execute_input":"2023-02-21T19:54:40.215687Z","iopub.status.idle":"2023-02-21T19:54:41.923132Z","shell.execute_reply.started":"2023-02-21T19:54:40.215651Z","shell.execute_reply":"2023-02-21T19:54:41.922088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Risk_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\nrisk_variables = pca_features[['Risk_{}'.format(i) for i in range(1, 15 )]]\nrisk_variables","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:54:48.932825Z","iopub.execute_input":"2023-02-21T19:54:48.933327Z","iopub.status.idle":"2023-02-21T19:54:49.840349Z","shell.execute_reply.started":"2023-02-21T19:54:48.933282Z","shell.execute_reply":"2023-02-21T19:54:49.839279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final data after Reduction\namex_data_red = pd.concat([deliquency_variables, spend_variables, payment_variables, balance_variables, risk_variables], axis = 1)\namex_data_processed = amex_data_red.join(amex_labels)\namex_data_processed.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:54:56.958411Z","iopub.execute_input":"2023-02-21T19:54:56.958983Z","iopub.status.idle":"2023-02-21T19:54:57.081954Z","shell.execute_reply.started":"2023-02-21T19:54:56.958933Z","shell.execute_reply":"2023-02-21T19:54:57.080734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split data into test and train","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:20:15.325847Z","iopub.execute_input":"2023-02-21T19:20:15.326239Z","iopub.status.idle":"2023-02-21T19:20:15.331299Z","shell.execute_reply.started":"2023-02-21T19:20:15.326204Z","shell.execute_reply":"2023-02-21T19:20:15.330370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pca data splitting\nx = amex_data_processed.loc[:, amex_data_processed.columns != 'target']\ny = amex_data_processed['target']\nx_train, x_test, y_train, y_test = train_test_split(x, y,stratify=y,test_size=0.20, random_state=10)\n\nprint('Train Data size: ',len(x_train))\nprint('Test Data size: ',len(x_test))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:01:39.819632Z","iopub.execute_input":"2023-02-21T20:01:39.820045Z","iopub.status.idle":"2023-02-21T20:01:39.914689Z","shell.execute_reply.started":"2023-02-21T20:01:39.820013Z","shell.execute_reply":"2023-02-21T20:01:39.913576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape\ny.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:01:42.894125Z","iopub.execute_input":"2023-02-21T20:01:42.894671Z","iopub.status.idle":"2023-02-21T20:01:42.906730Z","shell.execute_reply.started":"2023-02-21T20:01:42.894623Z","shell.execute_reply":"2023-02-21T20:01:42.905890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('x_train:', x_train.shape)\nprint('y_train:', y_train.shape)\nprint('x_test:', x_test.shape)\nprint('y_test:', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:02:09.694296Z","iopub.execute_input":"2023-02-21T20:02:09.695628Z","iopub.status.idle":"2023-02-21T20:02:09.701664Z","shell.execute_reply.started":"2023-02-21T20:02:09.695568Z","shell.execute_reply":"2023-02-21T20:02:09.700506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Normal Splitting of the data\nfrom sklearn.preprocessing import LabelEncoder\nlab_enc = LabelEncoder()\nfor cat_feat in cat_cols:\n    df_train[cat_feat] = lab_enc.fit_transform(df_train[cat_feat])","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:02:45.408746Z","iopub.execute_input":"2023-02-21T20:02:45.409163Z","iopub.status.idle":"2023-02-21T20:02:45.451678Z","shell.execute_reply.started":"2023-02-21T20:02:45.409131Z","shell.execute_reply":"2023-02-21T20:02:45.450594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop('target', axis=1)\nY = df_train['target']\nX.info()\nX.shape\nY.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:02:48.703357Z","iopub.execute_input":"2023-02-21T20:02:48.704038Z","iopub.status.idle":"2023-02-21T20:02:48.828718Z","shell.execute_reply.started":"2023-02-21T20:02:48.704000Z","shell.execute_reply":"2023-02-21T20:02:48.827217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test , Y_train , Y_test = train_test_split(X,Y,test_size=0.2,random_state=42) # 80-20 split\n\n# Checking split \nprint('X_train:', X_train.shape)\nprint('Y_train:', Y_train.shape)\nprint('X_test:', X_test.shape)\nprint('Y_test:', Y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:02:53.789712Z","iopub.execute_input":"2023-02-21T20:02:53.790190Z","iopub.status.idle":"2023-02-21T20:02:53.917210Z","shell.execute_reply.started":"2023-02-21T20:02:53.790151Z","shell.execute_reply":"2023-02-21T20:02:53.916060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CatBoostClassifier","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nclf = CatBoostClassifier(iterations = 3000, random_state = 42)\nclf.fit(X_train, Y_train, eval_set = [(X_test, Y_test)], cat_features=cat_cols,  verbose = 100)\npreds = clf.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:02:57.689416Z","iopub.execute_input":"2023-02-21T20:02:57.690572Z","iopub.status.idle":"2023-02-21T20:10:42.415039Z","shell.execute_reply.started":"2023-02-21T20:02:57.690531Z","shell.execute_reply":"2023-02-21T20:10:42.413440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nfrom sklearn.metrics import confusion_matrix \nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:10:42.418094Z","iopub.execute_input":"2023-02-21T20:10:42.418642Z","iopub.status.idle":"2023-02-21T20:10:42.425599Z","shell.execute_reply.started":"2023-02-21T20:10:42.418592Z","shell.execute_reply":"2023-02-21T20:10:42.423913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred = clf.predict(X_test) ","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:10:42.427887Z","iopub.execute_input":"2023-02-21T20:10:42.428369Z","iopub.status.idle":"2023-02-21T20:10:42.550470Z","shell.execute_reply.started":"2023-02-21T20:10:42.428332Z","shell.execute_reply":"2023-02-21T20:10:42.549073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_accuracy1 = clf.score(X_test, Y_test)\nprint(\"Test accuracy: {:.2f}%\".format(test_accuracy1 * 100))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:10:42.554177Z","iopub.execute_input":"2023-02-21T20:10:42.554714Z","iopub.status.idle":"2023-02-21T20:10:42.679377Z","shell.execute_reply.started":"2023-02-21T20:10:42.554643Z","shell.execute_reply":"2023-02-21T20:10:42.677963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(Y_test, Y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:10:42.681165Z","iopub.execute_input":"2023-02-21T20:10:42.682577Z","iopub.status.idle":"2023-02-21T20:10:42.700562Z","shell.execute_reply.started":"2023-02-21T20:10:42.682524Z","shell.execute_reply":"2023-02-21T20:10:42.699054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:10:42.702633Z","iopub.execute_input":"2023-02-21T20:10:42.703654Z","iopub.status.idle":"2023-02-21T20:10:42.713545Z","shell.execute_reply.started":"2023-02-21T20:10:42.703601Z","shell.execute_reply":"2023-02-21T20:10:42.711960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,7],)\nsns.heatmap(cm, annot = True)\nplt.show()\nprint(classification_report(Y_test, Y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:10:42.715294Z","iopub.execute_input":"2023-02-21T20:10:42.715661Z","iopub.status.idle":"2023-02-21T20:10:43.024406Z","shell.execute_reply.started":"2023-02-21T20:10:42.715630Z","shell.execute_reply":"2023-02-21T20:10:43.022982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.8\ndf_train = df_train.drop(df_train.columns[df_train.isnull().mean() >= threshold], axis=1)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:00.473872Z","iopub.execute_input":"2023-02-21T20:19:00.474426Z","iopub.status.idle":"2023-02-21T20:19:00.585440Z","shell.execute_reply.started":"2023-02-21T20:19:00.474379Z","shell.execute_reply":"2023-02-21T20:19:00.583832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:06.096031Z","iopub.execute_input":"2023-02-21T20:19:06.096452Z","iopub.status.idle":"2023-02-21T20:19:06.105296Z","shell.execute_reply.started":"2023-02-21T20:19:06.096421Z","shell.execute_reply":"2023-02-21T20:19:06.104225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:24.528926Z","iopub.execute_input":"2023-02-21T20:19:24.529390Z","iopub.status.idle":"2023-02-21T20:19:24.538734Z","shell.execute_reply.started":"2023-02-21T20:19:24.529351Z","shell.execute_reply":"2023-02-21T20:19:24.537102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:28.506339Z","iopub.execute_input":"2023-02-21T20:19:28.506811Z","iopub.status.idle":"2023-02-21T20:19:30.073875Z","shell.execute_reply.started":"2023-02-21T20:19:28.506775Z","shell.execute_reply":"2023-02-21T20:19:30.072397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:31.246489Z","iopub.execute_input":"2023-02-21T20:19:31.247018Z","iopub.status.idle":"2023-02-21T20:19:31.286439Z","shell.execute_reply.started":"2023-02-21T20:19:31.246981Z","shell.execute_reply":"2023-02-21T20:19:31.284999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:38.022351Z","iopub.execute_input":"2023-02-21T20:19:38.022840Z","iopub.status.idle":"2023-02-21T20:19:38.099924Z","shell.execute_reply.started":"2023-02-21T20:19:38.022806Z","shell.execute_reply":"2023-02-21T20:19:38.098610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop('target', axis=1)\ny = df_train['target']","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:35:02.620676Z","iopub.execute_input":"2023-02-21T20:35:02.621261Z","iopub.status.idle":"2023-02-21T20:35:02.772843Z","shell.execute_reply.started":"2023-02-21T20:35:02.621221Z","shell.execute_reply":"2023-02-21T20:35:02.771637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:44:19.719067Z","iopub.execute_input":"2023-02-21T20:44:19.719561Z","iopub.status.idle":"2023-02-21T20:44:19.727314Z","shell.execute_reply.started":"2023-02-21T20:44:19.719523Z","shell.execute_reply":"2023-02-21T20:44:19.725939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesClassifier\nimport matplotlib.pyplot as plt\nmodel = ExtraTreesClassifier()\nmodel.fit(X,y)\nprint(model.feature_importances_)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:44:23.159457Z","iopub.execute_input":"2023-02-21T20:44:23.159897Z","iopub.status.idle":"2023-02-21T20:44:58.041400Z","shell.execute_reply.started":"2023-02-21T20:44:23.159866Z","shell.execute_reply":"2023-02-21T20:44:58.040258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.iloc[:,:-1]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T21:23:23.621062Z","iopub.execute_input":"2023-02-21T21:23:23.621573Z","iopub.status.idle":"2023-02-21T21:23:23.675183Z","shell.execute_reply.started":"2023-02-21T21:23:23.621535Z","shell.execute_reply":"2023-02-21T21:23:23.673749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_importances = pd.Series(model.feature_importances_, index=X.columns)\nfeat_importances.nlargest(10).plot(kind='barh')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T21:23:44.062174Z","iopub.execute_input":"2023-02-21T21:23:44.062590Z","iopub.status.idle":"2023-02-21T21:23:44.343713Z","shell.execute_reply.started":"2023-02-21T21:23:44.062559Z","shell.execute_reply":"2023-02-21T21:23:44.342597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Catbootclassification with reduced data\nfrom catboost import CatBoostClassifier\nclfl = CatBoostClassifier(iterations = 3000, random_state = 42, loss_function='Logloss')\nclfl.fit(x_train, y_train, eval_set = [(x_test, y_test)],  verbose = 100)\npreds = clfl.predict_proba(x_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:19:41.432518Z","iopub.execute_input":"2023-02-21T20:19:41.433564Z","iopub.status.idle":"2023-02-21T20:21:30.441219Z","shell.execute_reply.started":"2023-02-21T20:19:41.433514Z","shell.execute_reply":"2023-02-21T20:21:30.438401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clfl.predict(x_test) ","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:30:35.205683Z","iopub.execute_input":"2023-02-21T19:30:35.207558Z","iopub.status.idle":"2023-02-21T19:30:35.252642Z","shell.execute_reply.started":"2023-02-21T19:30:35.207491Z","shell.execute_reply":"2023-02-21T19:30:35.250650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_accuracy = clfl.score(x_test, y_test)\nprint(\"Test accuracy: {:.2f}%\".format(test_accuracy * 100))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:30:35.254905Z","iopub.execute_input":"2023-02-21T19:30:35.255376Z","iopub.status.idle":"2023-02-21T19:30:35.299282Z","shell.execute_reply.started":"2023-02-21T19:30:35.255340Z","shell.execute_reply":"2023-02-21T19:30:35.298147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm1 = confusion_matrix(Y_test, Y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:30:35.301286Z","iopub.execute_input":"2023-02-21T19:30:35.302485Z","iopub.status.idle":"2023-02-21T19:30:35.311530Z","shell.execute_reply.started":"2023-02-21T19:30:35.302442Z","shell.execute_reply":"2023-02-21T19:30:35.310307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm1","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:30:35.314056Z","iopub.execute_input":"2023-02-21T19:30:35.314547Z","iopub.status.idle":"2023-02-21T19:30:35.333447Z","shell.execute_reply.started":"2023-02-21T19:30:35.314506Z","shell.execute_reply":"2023-02-21T19:30:35.330724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,7],)\nsns.heatmap(cm1, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:30:35.336140Z","iopub.execute_input":"2023-02-21T19:30:35.336779Z","iopub.status.idle":"2023-02-21T19:30:35.646743Z","shell.execute_reply.started":"2023-02-21T19:30:35.336720Z","shell.execute_reply":"2023-02-21T19:30:35.645296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T19:36:09.212185Z","iopub.execute_input":"2023-02-21T19:36:09.213821Z","iopub.status.idle":"2023-02-21T19:36:09.277346Z","shell.execute_reply.started":"2023-02-21T19:36:09.213751Z","shell.execute_reply":"2023-02-21T19:36:09.275454Z"},"trusted":true},"execution_count":null,"outputs":[]}]}