{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load Libraries","metadata":{"id":"L4qOvl35eWG7"}},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd\nimport numpy as np \nimport sklearn\nimport plotnine \nfrom plotnine import *\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pickle","metadata":{"id":"Usk_vuJ_eWHA","outputId":"972d687f-6ca8-42a5-def9-a88cbe807f77","execution":{"iopub.status.busy":"2022-06-29T21:00:30.379327Z","iopub.execute_input":"2022-06-29T21:00:30.37977Z","iopub.status.idle":"2022-06-29T21:00:33.303586Z","shell.execute_reply.started":"2022-06-29T21:00:30.379731Z","shell.execute_reply":"2022-06-29T21:00:33.302524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process and Feature Engineer Train Data\n","metadata":{"id":"HgSVABg0eWHD"}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = pd.read_parquet(path, columns=usecols)\n    else: df = pd.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].apply(lambda x: int(x[-16:],16)).astype('int64')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    df = df.sort_values(['customer_ID','S_2'])\n    df = df.reset_index(drop=True)\n    # FILL NAN\n    #df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = TRAIN_PATH)","metadata":{"id":"W3LcsLDOeWHE","outputId":"f35b7a61-22a9-4ceb-ea2f-30ebb168a84d","execution":{"iopub.status.busy":"2022-06-29T21:00:33.532712Z","iopub.execute_input":"2022-06-29T21:00:33.533083Z","iopub.status.idle":"2022-06-29T21:01:08.744761Z","shell.execute_reply.started":"2022-06-29T21:00:33.533054Z","shell.execute_reply":"2022-06-29T21:01:08.743353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntarget_path = '../input/amex-default-prediction/train_labels.csv'\n\ntarget = pd.read_csv(target_path)\ntarget['customer_ID'] = target['customer_ID'].apply(lambda x: int(x[-16:],16)).astype('int64')\n\ntarget['target'] = target.target.astype('int8')","metadata":{"id":"kmZgmgJkeWHF","execution":{"iopub.status.busy":"2022-06-29T21:01:08.747142Z","iopub.execute_input":"2022-06-29T21:01:08.747508Z","iopub.status.idle":"2022-06-29T21:01:10.515757Z","shell.execute_reply.started":"2022-06-29T21:01:08.747467Z","shell.execute_reply":"2022-06-29T21:01:10.514582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****","metadata":{"id":"qWQ123DSeWHG"}},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"id":"TON45vNoeWHH","outputId":"c3ad7021-410b-46cf-afe3-1c6d24497c4a","execution":{"iopub.status.busy":"2022-06-29T21:01:10.522116Z","iopub.execute_input":"2022-06-29T21:01:10.524561Z","iopub.status.idle":"2022-06-29T21:01:10.769713Z","shell.execute_reply.started":"2022-06-29T21:01:10.524328Z","shell.execute_reply":"2022-06-29T21:01:10.768205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering","metadata":{"id":"Ek0GNl8meWHI"}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\n\n# \nclass aggregation_FeatEngin(BaseEstimator, TransformerMixin) :\n    \n    def fit(self, X, y= None) :\n        return self\n    \n    def transform(self, X, y = None) : \n        print('Aggregation is processing...')\n        all_cols = [c for c in list(X.columns) if c not in ['customer_ID','S_2']]\n        cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n        num_features = [col for col in all_cols if col not in cat_features]\n\n        num_agg = X.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n        num_agg.columns = ['_'.join(x) for x in num_agg.columns]\n\n        cat_agg = X.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n        cat_agg.columns = ['_'.join(x) for x in cat_agg.columns]\n\n        X = pd.concat([num_agg, cat_agg], axis=1)\n        del num_agg, cat_agg\n    \n        return X\n\n# fillna for std cols :\nclass fillna_std_cols(BaseEstimator, TransformerMixin) :\n        \n    def fit(self, X, y= None) :\n        return self\n    \n    def transform(self, X, y = None) :\n        print('Std fillna is processing...')\n        std_col = [col for col in X.columns if col.endswith('std')]\n        X[std_col] = X[std_col].fillna(0)\n        return X\n\n\n# fillna float and cats\n\nclass fillna_RestOfCols(BaseEstimator, TransformerMixin) :\n        \n    def fit(self, X, y= None) :\n        return self\n    \n    def transform(self, X, y = None) :\n        print('RestOfCols fillna is processing...')\n        all_cols = [c for c in list(X.columns) if c not in ['customer_ID','S_2']]\n        cats = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n        cat_features = []\n        metrics = ['count', 'last', 'nunique']\n        for i in metrics :\n            cat_features.append([col + '_' + i for col in cats])\n        cat_features = [element for lst in cat_features for element in lst]\n        num_features = [col for col in all_cols if col not in cat_features]\n        X[num_features] = X[num_features].fillna(X[num_features].median())\n        X[cat_features] = X[cat_features].fillna(X[cat_features].mode())                                         \n        return X\n\nclass OneHotEnc_np(BaseEstimator, TransformerMixin) :\n\n    def fit(self, X, y= None) :\n        return self\n  \n    def to_one_hot(self, x):\n        return np.diag(np.ones(x.max() + 1))[x]\n    \n    def transform(self, X, y = None) :\n        print('OneHotEncoding is processing...')\n        cats = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n        cat_features = [col + '_' + 'last' for col in cats]\n        cat_isin = np.isin(X.columns.to_numpy(), np.array(cat_features))\n        cats_idx = np.arange(0, X.shape[1])[cat_isin]\n        \n        X = X.values\n        m = X.shape[0] \n        cats_enc = np.empty((m, 1))     \n        for id in cats_idx : \n            col_enc = self.to_one_hot(X[:, id].astype('int8'))\n            cats_enc = np.concatenate((cats_enc, col_enc), axis = 1)\n        cats_enc = np.apply_along_axis(lambda x : x.astype('float16'), arr = cats_enc, axis = 1)\n        print('Scaling is processing...')\n        \n        return np.concatenate((np.delete(X, cats_idx, axis = 1), cats_enc), axis = 1)","metadata":{"id":"LxyQY-D1eWHJ","execution":{"iopub.status.busy":"2022-06-29T21:01:10.772574Z","iopub.execute_input":"2022-06-29T21:01:10.773089Z","iopub.status.idle":"2022-06-29T21:01:10.793381Z","shell.execute_reply.started":"2022-06-29T21:01:10.773057Z","shell.execute_reply":"2022-06-29T21:01:10.792143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_eng_pipepline \n\ndef train_feat_eng_pipeline(X) :\n    agg_fe = aggregation_FeatEngin()\n    fillna_std = fillna_std_cols()\n    fillna_Rest = fillna_RestOfCols()\n    cat_encoder = sklearn.preprocessing.OneHotEncoder(sparse=False)\n    scaler = sklearn.preprocessing.StandardScaler()\n    \n    # Agg + fill na\n    X = agg_fe.fit_transform(X)\n    features_name = X.columns \n    X = fillna_std.fit_transform(X)\n    X = fillna_Rest.fit_transform(X)\n    \n    # Handling cat vars (OneHotEncod)\n    cats = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    cat_features = [col + '_' + 'last' for col in cats]\n    X_cat = X[cat_features]\n    X_encod = cat_encoder.fit_transform(X_cat)\n    del X_cat\n    cat_cols = ['cat' + '_' + str(i) for i in np.arange(1, X_encod.shape[1] + 1)]\n    cat_df = pd.DataFrame(X_encod)\n    del X_encod\n    cat_df.columns = cat_cols\n    anti_cols = [col for col in X.columns if col not in cat_features]\n    X = pd.merge(X[anti_cols].reset_index(), cat_df,left_index= True, right_index= True, how = 'outer')\n    \n    # scaling \n    scale_cols = [col for col in X.columns if col not in cat_cols + ['customer_ID']]\n    X[scale_cols] = scaler.fit_transform(X[scale_cols])\n    \n    return X","metadata":{"id":"X2fkGUUDeWHL","execution":{"iopub.status.busy":"2022-06-29T21:01:10.794984Z","iopub.execute_input":"2022-06-29T21:01:10.795409Z","iopub.status.idle":"2022-06-29T21:01:10.814149Z","shell.execute_reply.started":"2022-06-29T21:01:10.795368Z","shell.execute_reply":"2022-06-29T21:01:10.813113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train_feat_eng_pipeline(train)\ngc.collect()","metadata":{"id":"XVudjRlmeWHM","outputId":"cd3167cc-18fe-4bbd-a802-7cbd8e2808c1","execution":{"iopub.status.busy":"2022-06-29T21:01:10.815755Z","iopub.execute_input":"2022-06-29T21:01:10.816713Z","iopub.status.idle":"2022-06-29T21:03:37.432923Z","shell.execute_reply.started":"2022-06-29T21:01:10.816667Z","shell.execute_reply":"2022-06-29T21:03:37.43167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## fit PCA","metadata":{"id":"3GURXu7FeWHO"}},{"cell_type":"code","source":"# y_train\nmap_target = dict(target.values)\ntarget = train['customer_ID'].map(map_target)","metadata":{"id":"aHBWKriQeWHP","execution":{"iopub.status.busy":"2022-06-29T21:03:37.434634Z","iopub.execute_input":"2022-06-29T21:03:37.434943Z","iopub.status.idle":"2022-06-29T21:03:38.447462Z","shell.execute_reply.started":"2022-06-29T21:03:37.434912Z","shell.execute_reply":"2022-06-29T21:03:38.44582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedShuffleSplit\n\nSSS = StratifiedShuffleSplit(n_splits= 1, train_size = 0.25, random_state= 7)\nfor train_idx, test_idx in SSS.split(train, target.values) : \n    train_sample = train.iloc[train_idx, :]\n    y = target.iloc[train_idx]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:03:38.449037Z","iopub.execute_input":"2022-06-29T21:03:38.449484Z","iopub.status.idle":"2022-06-29T21:03:42.158833Z","shell.execute_reply.started":"2022-06-29T21:03:38.449422Z","shell.execute_reply":"2022-06-29T21:03:42.157683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# perform PCA \n# We don't have to prescale our data because we already perform it within the pipeline function \nX = train_sample.iloc[:, 1:]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:03:42.160245Z","iopub.execute_input":"2022-06-29T21:03:42.160599Z","iopub.status.idle":"2022-06-29T21:03:42.467206Z","shell.execute_reply.started":"2022-06-29T21:03:42.160566Z","shell.execute_reply":"2022-06-29T21:03:42.466004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, train_sample \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:03:42.469962Z","iopub.execute_input":"2022-06-29T21:03:42.470386Z","iopub.status.idle":"2022-06-29T21:03:42.622264Z","shell.execute_reply.started":"2022-06-29T21:03:42.470354Z","shell.execute_reply":"2022-06-29T21:03:42.621085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca = PCA()\namex_pca = pca.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:03:42.623842Z","iopub.execute_input":"2022-06-29T21:03:42.624175Z","iopub.status.idle":"2022-06-29T21:04:03.109028Z","shell.execute_reply.started":"2022-06-29T21:03:42.624143Z","shell.execute_reply":"2022-06-29T21:04:03.108018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# singular_values exploration and varaibilité ratio \npca_var = pd.DataFrame()\npca_var['sing_val'] = pca.singular_values_\npca_var['var_exp'] = pca.explained_variance_\npca_var['var_exp_ratio'] = pca.explained_variance_ratio_\npca_var['var_cum'] = pca_var.var_exp_ratio.cumsum()\npca_var.index = np.arange(1,len(pca_var)+1)\npca_var = pca_var.reset_index()\npca_var.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:04:03.110823Z","iopub.execute_input":"2022-06-29T21:04:03.111238Z","iopub.status.idle":"2022-06-29T21:04:03.136972Z","shell.execute_reply.started":"2022-06-29T21:04:03.111196Z","shell.execute_reply":"2022-06-29T21:04:03.136222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# singular  value\n(\n    ggplot(pca_var, aes(x = 'sing_val'))\n    + geom_density(alpha = 0.7, fill = 'blue')\n    + theme_minimal()\n    + theme(figure_size = (4, 3))\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:45:55.801439Z","iopub.execute_input":"2022-06-29T21:45:55.801990Z","iopub.status.idle":"2022-06-29T21:45:56.188314Z","shell.execute_reply.started":"2022-06-29T21:45:55.801940Z","shell.execute_reply":"2022-06-29T21:45:56.187116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# variability distrib\n(\n    ggplot(pca_var, aes(x = 'index', y = 'var_cum'))\n    + geom_col(fill = 'orange')\n    + theme_minimal()\n    + theme(figure_size = (5, 4))\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:45:58.474711Z","iopub.execute_input":"2022-06-29T21:45:58.475089Z","iopub.status.idle":"2022-06-29T21:46:00.765460Z","shell.execute_reply.started":"2022-06-29T21:45:58.475058Z","shell.execute_reply":"2022-06-29T21:46:00.764150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Features contribution to synthetic axis. \neigen = pd.DataFrame(pca.components_, \n                     columns = pca.feature_names_in_, \n                     index = [f'PC{n + 1}' for n in range(pca.components_.shape[0])])\neigen.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:13.934574Z","iopub.execute_input":"2022-06-29T21:46:13.934987Z","iopub.status.idle":"2022-06-29T21:46:13.963248Z","shell.execute_reply.started":"2022-06-29T21:46:13.934952Z","shell.execute_reply":"2022-06-29T21:46:13.962507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reorganize columns by categories \npayement = [col for col in eigen.columns if col[0] == 'P']\nbalance = [col for col in eigen.columns if col[0] == 'B']\nspend = [col for col in eigen.columns if col[0] == 'S']\ndelin = [col for col in eigen.columns if col[0] == 'D']\nrisque = [col for col in eigen.columns if col[0] == 'R']\ncats = [col for col in eigen.columns if col[0] == 'c']\n\neigen = eigen[payement + spend + balance + delin + risque + cats]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:16.522277Z","iopub.execute_input":"2022-06-29T21:46:16.523031Z","iopub.status.idle":"2022-06-29T21:46:16.536628Z","shell.execute_reply.started":"2022-06-29T21:46:16.522989Z","shell.execute_reply":"2022-06-29T21:46:16.535573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# heat map ten first axis? \nplt.subplots(figsize=(20,7))  \nsns.heatmap(eigen.iloc[:6],cmap=\"PiYG\")","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:19.163426Z","iopub.execute_input":"2022-06-29T21:46:19.163809Z","iopub.status.idle":"2022-06-29T21:46:20.615004Z","shell.execute_reply.started":"2022-06-29T21:46:19.163777Z","shell.execute_reply":"2022-06-29T21:46:20.613742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We gonna keep just the 6 first axis : 25% of variability. \neigen = eigen[:6]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:23.700837Z","iopub.execute_input":"2022-06-29T21:46:23.701821Z","iopub.status.idle":"2022-06-29T21:46:23.708279Z","shell.execute_reply.started":"2022-06-29T21:46:23.701769Z","shell.execute_reply":"2022-06-29T21:46:23.706995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eigen = eigen.T.unstack().reset_index()\neigen.columns = ['comp', 'feature', 'weight']","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:25.857349Z","iopub.execute_input":"2022-06-29T21:46:25.858471Z","iopub.status.idle":"2022-06-29T21:46:25.867973Z","shell.execute_reply.started":"2022-06-29T21:46:25.858416Z","shell.execute_reply":"2022-06-29T21:46:25.866878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat = []\nfor i in range(300) :\n    cat.append('last')\n\nnon_cat = list(eigen['feature'][~ eigen.feature.str.startswith('cat')].str.split('_').apply(lambda x : x[2]))\n\neigen['category'] = eigen['feature'].str.split('_').apply(lambda x : x[0])\neigen['metric'] = non_cat + cat\neigen = eigen[['comp', 'feature', 'category', 'metric', 'weight']]\neigen['weight_abs'] = eigen.weight.abs()\neigen.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:28.220881Z","iopub.execute_input":"2022-06-29T21:46:28.221277Z","iopub.status.idle":"2022-06-29T21:46:28.259812Z","shell.execute_reply.started":"2022-06-29T21:46:28.221242Z","shell.execute_reply":"2022-06-29T21:46:28.258861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Category of variable contribution to the axis  \n(\n    ggplot(eigen, aes(x = 'category', y = 'weight_abs', fill = 'category'))\n    + geom_boxplot()\n    + facet_wrap('~ comp')\n    + theme(figure_size = (12, 5))\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:30.852686Z","iopub.execute_input":"2022-06-29T21:46:30.853079Z","iopub.status.idle":"2022-06-29T21:46:35.741860Z","shell.execute_reply.started":"2022-06-29T21:46:30.853049Z","shell.execute_reply":"2022-06-29T21:46:35.740782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(\n    ggplot(eigen, aes(x = 'metric', y = 'weight_abs', fill = 'metric'))\n    + geom_boxplot()\n    + facet_wrap('~ comp')\n    + theme(figure_size = (12, 5))\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:35.743363Z","iopub.execute_input":"2022-06-29T21:46:35.744071Z","iopub.status.idle":"2022-06-29T21:46:40.428938Z","shell.execute_reply.started":"2022-06-29T21:46:35.744034Z","shell.execute_reply":"2022-06-29T21:46:40.427858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# individual feature contribution to principal componenents\neigen_top10 = eigen.sort_values('weight_abs', ascending = False).groupby('comp').head(10)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:40.430715Z","iopub.execute_input":"2022-06-29T21:46:40.431050Z","iopub.status.idle":"2022-06-29T21:46:40.440563Z","shell.execute_reply.started":"2022-06-29T21:46:40.431018Z","shell.execute_reply":"2022-06-29T21:46:40.439636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(\n    ggplot(eigen_top10, aes(x = 'feature', y = 'weight', fill = 'feature'))\n    + geom_col(alpha = 0.8, show_legend = False)\n    + facet_wrap('~ comp', scales = 'free_y')\n    + coord_flip()\n    + theme_minimal()\n    + theme(figure_size = (12, 4),\n           axis_text_y= element_text(size = 8))\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:47.992370Z","iopub.execute_input":"2022-06-29T21:46:47.993806Z","iopub.status.idle":"2022-06-29T21:46:49.237758Z","shell.execute_reply.started":"2022-06-29T21:46:47.993761Z","shell.execute_reply":"2022-06-29T21:46:49.236689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation matrix of the top 10 \ntop10 = list(eigen_top10.feature.value_counts().index)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:57:36.705101Z","iopub.execute_input":"2022-06-29T21:57:36.705498Z","iopub.status.idle":"2022-06-29T21:57:36.711842Z","shell.execute_reply.started":"2022-06-29T21:57:36.705465Z","shell.execute_reply":"2022-06-29T21:57:36.710541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"payement = [col for col in top10 if col[0] == 'P']\nbalance = [col for col in top10 if col[0] == 'B']\nspend = [col for col in top10 if col[0] == 'S']\ndelin = [col for col in top10 if col[0] == 'D']\nrisque = [col for col in top10 if col[0] == 'R']\ncats = [col for col in top10 if col[0] == 'c']","metadata":{"execution":{"iopub.status.busy":"2022-06-29T22:07:46.209555Z","iopub.execute_input":"2022-06-29T22:07:46.210027Z","iopub.status.idle":"2022-06-29T22:07:46.217601Z","shell.execute_reply.started":"2022-06-29T22:07:46.209993Z","shell.execute_reply":"2022-06-29T22:07:46.216362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = X[payement + spend + balance + delin + risque + cats].corr()\nmatrix = np.triu(corr)\nplt.subplots(figsize=(13,13))  \nsns.heatmap(corr, mask = matrix, cmap=\"PiYG\")","metadata":{"execution":{"iopub.status.busy":"2022-06-29T22:14:45.104552Z","iopub.execute_input":"2022-06-29T22:14:45.104963Z","iopub.status.idle":"2022-06-29T22:14:47.636849Z","shell.execute_reply.started":"2022-06-29T22:14:45.104929Z","shell.execute_reply":"2022-06-29T22:14:47.635725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Projection of rows onto pca axis \nnew_amex = pd.DataFrame()\nnew_amex['x1'] = amex_pca[:, 0]\nnew_amex['x2'] = amex_pca[:, 1]\nnew_amex['x3'] = amex_pca[:, 2]\nnew_amex['x4'] = amex_pca[:, 3]\nnew_amex['x5'] = amex_pca[:, 4]\nnew_amex['x6'] = amex_pca[:, 5]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:52.476903Z","iopub.execute_input":"2022-06-29T21:46:52.477831Z","iopub.status.idle":"2022-06-29T21:46:52.497808Z","shell.execute_reply.started":"2022-06-29T21:46:52.477783Z","shell.execute_reply":"2022-06-29T21:46:52.496864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x1 ~ x2 : 14% var explainned \n(\n    ggplot(new_amex, aes(x = 'x1', y = 'x2', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:46:54.948959Z","iopub.execute_input":"2022-06-29T21:46:54.949772Z","iopub.status.idle":"2022-06-29T21:47:01.692798Z","shell.execute_reply.started":"2022-06-29T21:46:54.949730Z","shell.execute_reply":"2022-06-29T21:47:01.691733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x1 ~ x3 : 12% var explainned \n(\n    ggplot(new_amex, aes(x = 'x1', y = 'x3', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:01.695278Z","iopub.execute_input":"2022-06-29T21:47:01.695979Z","iopub.status.idle":"2022-06-29T21:47:08.378461Z","shell.execute_reply.started":"2022-06-29T21:47:01.695937Z","shell.execute_reply":"2022-06-29T21:47:08.377379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x1 ~ x4 : 12% var explainned \n(\n    ggplot(new_amex, aes(x = 'x1', y = 'x4', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:08.379940Z","iopub.execute_input":"2022-06-29T21:47:08.380253Z","iopub.status.idle":"2022-06-29T21:47:15.093384Z","shell.execute_reply.started":"2022-06-29T21:47:08.380225Z","shell.execute_reply":"2022-06-29T21:47:15.092291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x1 ~ x5\n(\n    ggplot(new_amex, aes(x = 'x1', y = 'x5', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:15.095877Z","iopub.execute_input":"2022-06-29T21:47:15.096295Z","iopub.status.idle":"2022-06-29T21:47:21.738600Z","shell.execute_reply.started":"2022-06-29T21:47:15.096264Z","shell.execute_reply":"2022-06-29T21:47:21.737866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x2 ~ x3 \n(\n    ggplot(new_amex, aes(x = 'x2', y = 'x3', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:21.739800Z","iopub.execute_input":"2022-06-29T21:47:21.740543Z","iopub.status.idle":"2022-06-29T21:47:28.514103Z","shell.execute_reply.started":"2022-06-29T21:47:21.740513Z","shell.execute_reply":"2022-06-29T21:47:28.512929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x2 ~ x4 \n(\n    ggplot(new_amex, aes(x = 'x2', y = 'x4', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:28.515352Z","iopub.execute_input":"2022-06-29T21:47:28.515997Z","iopub.status.idle":"2022-06-29T21:47:35.203638Z","shell.execute_reply.started":"2022-06-29T21:47:28.515959Z","shell.execute_reply":"2022-06-29T21:47:35.202523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x2 ~ x5 \n(\n    ggplot(new_amex, aes(x = 'x2', y = 'x5', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:35.204793Z","iopub.execute_input":"2022-06-29T21:47:35.205101Z","iopub.status.idle":"2022-06-29T21:47:41.870653Z","shell.execute_reply.started":"2022-06-29T21:47:35.205071Z","shell.execute_reply":"2022-06-29T21:47:41.869534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x3 ~ x4 \n(\n    ggplot(new_amex, aes(x = 'x3', y = 'x4', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:41.872559Z","iopub.execute_input":"2022-06-29T21:47:41.873015Z","iopub.status.idle":"2022-06-29T21:47:48.594991Z","shell.execute_reply.started":"2022-06-29T21:47:41.872970Z","shell.execute_reply":"2022-06-29T21:47:48.593921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x3 ~ x5\n(\n    ggplot(new_amex, aes(x = 'x3', y = 'x5', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:48.596585Z","iopub.execute_input":"2022-06-29T21:47:48.597018Z","iopub.status.idle":"2022-06-29T21:47:55.236275Z","shell.execute_reply.started":"2022-06-29T21:47:48.596974Z","shell.execute_reply":"2022-06-29T21:47:55.235072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x4 ~ x5\n(\n    ggplot(new_amex, aes(x = 'x4', y = 'x5', fill = 'factor(y)'))\n    + geom_point(alpha = 0.7, size = 2)\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T21:47:55.250066Z","iopub.execute_input":"2022-06-29T21:47:55.251141Z","iopub.status.idle":"2022-06-29T21:48:02.064010Z","shell.execute_reply.started":"2022-06-29T21:47:55.251097Z","shell.execute_reply":"2022-06-29T21:48:02.062914Z"},"trusted":true},"execution_count":null,"outputs":[]}]}