{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ====================================================\n# Library\n# ====================================================\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\nfrom tqdm.auto import tqdm\nfrom scipy import stats","metadata":{"id":"8IcxJd_WOjhF","execution":{"iopub.status.busy":"2022-07-22T03:35:05.561618Z","iopub.execute_input":"2022-07-22T03:35:05.562106Z","iopub.status.idle":"2022-07-22T03:35:06.194147Z","shell.execute_reply.started":"2022-07-22T03:35:05.562064Z","shell.execute_reply":"2022-07-22T03:35:06.192860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check stable Independent Variable in train and test period","metadata":{"id":"kb8xfxjpVgfU"}},{"cell_type":"markdown","source":"### Characteristic Stability Index (CSI)\n\nThe Characteristic Stability Index (CSI) is used to evaluate the stability or drift of each feature so that we can find the problematic one. As PSI is concerned with the effects of the population drift on the model’s predictions, the CSI is concerned with understanding how the feature distributions have changed\n\n- CSI < 0.1 = The variable  hasn’t changed, and we can use to train the model\n- 0.1 ≤ CS1 < 0.2 = The variable  has slightly changed, and it is advisable to evaluate the impacts of these changes\n- CSI ≥ 0.2 = The changes in variable  are significant, and the model should not be used the characteristic in model.\n\n","metadata":{"id":"Q-YM3AN2YalM"}},{"cell_type":"markdown","source":"### Kolmogorov–Smirnov method (K–S test) \nThe Kolmogorov–Smirnov method (K–S test) is used to compare the maximum distance between the experimental cumulative distribution function and the theoretical cumulative distribution function. A more general approach is to test for differences in the entire two distributions, e.g. training features vs test features.\n","metadata":{}},{"cell_type":"code","source":"def ks_test():\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\n    binary_features = ['B_31', 'D_87']\n\n    cat_features = cat_features + binary_features\n    statistic = []\n    pvalue = []    \n\n    num_features = list(set(train.columns) - set(cat_features))\n    # num_features = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)    \n    for i, name in enumerate(num_features):\n        statistic_, pvalue_ = stats.ks_2samp(train[name], test[name])\n        statistic.append(statistic_)\n        pvalue.append(pvalue_) \n    return pd.DataFrame({'name':num_features, 'ks':statistic, 'pvalue':pvalue})\n\ndef csi(var):    \n    x = train.groupby(var).size().to_frame()\n    x.reset_index(inplace = True)        \n    y = test.groupby(var).size().to_frame()\n    y.reset_index(inplace = True)    \n    csi_tbl = x.merge(y, how = 'inner', on = var)\n    csi_tbl['perc_train'] = csi_tbl['0_x']/sum(csi_tbl['0_x'])\n    csi_tbl['perc_test']= csi_tbl['0_y']/sum(csi_tbl['0_y'])\n    csi_tbl['csi_sub']= (csi_tbl['perc_train']-csi_tbl['perc_test']) * np.log(csi_tbl['perc_train']/csi_tbl['perc_test'])    \n    return sum(csi_tbl['csi_sub'])\n\ndef csi_test():\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\n    binary_features = ['B_31', 'D_87']\n    cat_features = cat_features + binary_features\n    csi_out = []\n    for i, name in enumerate(cat_features):        \n        csi_out.append(csi(name))                \n    return pd.DataFrame({'name':cat_features, 'csi':csi_out})","metadata":{"id":"NartAGBtV2cq","execution":{"iopub.status.busy":"2022-07-22T03:38:17.131622Z","iopub.execute_input":"2022-07-22T03:38:17.132067Z","iopub.status.idle":"2022-07-22T03:38:17.147963Z","shell.execute_reply.started":"2022-07-22T03:38:17.132031Z","shell.execute_reply":"2022-07-22T03:38:17.146719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\ntest = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')","metadata":{"id":"YtVmkB1kibBi","execution":{"iopub.status.busy":"2022-07-22T03:35:22.554106Z","iopub.execute_input":"2022-07-22T03:35:22.554556Z","iopub.status.idle":"2022-07-22T03:36:18.687055Z","shell.execute_reply.started":"2022-07-22T03:35:22.554521Z","shell.execute_reply":"2022-07-22T03:36:18.685759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csi_table = csi_test().sort_values('csi')\ncsi_table","metadata":{"id":"gQGyGrhMjhq2","execution":{"iopub.status.busy":"2022-07-22T03:38:21.850630Z","iopub.execute_input":"2022-07-22T03:38:21.851051Z","iopub.status.idle":"2022-07-22T03:38:26.938051Z","shell.execute_reply.started":"2022-07-22T03:38:21.851013Z","shell.execute_reply":"2022-07-22T03:38:26.936712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"==> Category variables are stable between train and test period\n","metadata":{}},{"cell_type":"code","source":"ks_table = ks_test().sort_values('ks')\nks_table","metadata":{"id":"JojFEKI_iVD5","execution":{"iopub.status.busy":"2022-07-22T03:38:31.967893Z","iopub.execute_input":"2022-07-22T03:38:31.968346Z","iopub.status.idle":"2022-07-22T03:49:53.685770Z","shell.execute_reply.started":"2022-07-22T03:38:31.968310Z","shell.execute_reply":"2022-07-22T03:49:53.683127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"S_9, B_29, D_59, S_11, R1, S_2 should be careful to using in training model.\n","metadata":{}}]}