{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Adversarial Validation\nIn this Notebook, Adversarial Validation was performed on Train data and Public data, and on Train data and Private data, and the differences were compared.  \nThe correlation between CV and LB is stable, so there is no need to worry too much. However, Public and Private are divided in time series, so we need to be very careful to prevent Shake Down.  ","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"import cudf\nimport cupy\nimport pandas as pd\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nfrom catboost import Pool\nfrom catboost import CatBoost\nimport numpy as np\nfrom tqdm import tqdm\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport gc\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-14T20:56:54.048216Z","iopub.execute_input":"2022-06-14T20:56:54.048620Z","iopub.status.idle":"2022-06-14T20:56:57.822582Z","shell.execute_reply.started":"2022-06-14T20:56:54.048542Z","shell.execute_reply":"2022-06-14T20:56:57.821734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"test = cudf.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')\ntest = test.drop_duplicates(subset=[\"customer_ID\"], keep=\"last\")\ntest['S_2'] = cudf.to_datetime(test['S_2'])\ntest['month'] = (test['S_2'].dt.month).astype('int8')\n\ntrain = cudf.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\ntrain = train.drop_duplicates(subset=[\"customer_ID\"], keep=\"last\")\ntrain['S_2'] = cudf.to_datetime(train['S_2'])\ntrain['month'] = (train['S_2'].dt.month).astype('int8')","metadata":{"execution":{"iopub.status.busy":"2022-06-14T20:58:12.728939Z","iopub.execute_input":"2022-06-14T20:58:12.729481Z","iopub.status.idle":"2022-06-14T20:59:19.783766Z","shell.execute_reply.started":"2022-06-14T20:58:12.729449Z","shell.execute_reply":"2022-06-14T20:59:19.782781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'] = 1\ntest['target'] = 0\ntest['target'] = 0\n\ntest_private = test[test['month'] == 4].reset_index(drop=True)\ntest_public = test[test['month'] == 10].reset_index(drop=True)\ndel test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T20:59:19.785351Z","iopub.execute_input":"2022-06-14T20:59:19.785688Z","iopub.status.idle":"2022-06-14T20:59:21.205340Z","shell.execute_reply.started":"2022-06-14T20:59:19.785655Z","shell.execute_reply":"2022-06-14T20:59:21.204584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train vs Public\nIn this case, Catboost (GPU) is used.  \nYou may use feature importance such as Lightgbm for comparison.  \nAlso, to save time, we stopped running at AUC<0.75, but we recommend continuing until the AUC is less than 0.6.","metadata":{}},{"cell_type":"code","source":"data = cudf.concat([train,test_public]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:02:22.922219Z","iopub.execute_input":"2022-06-14T21:02:22.922752Z","iopub.status.idle":"2022-06-14T21:02:23.056570Z","shell.execute_reply.started":"2022-06-14T21:02:22.922718Z","shell.execute_reply":"2022-06-14T21:02:23.055650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET = 'target'\ndrop_cols = ['S_2','month','customer_ID',TARGET]\nuse_cols = [c for c in data.columns if c not in drop_cols]","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:02:25.551023Z","iopub.execute_input":"2022-06-14T21:02:25.551726Z","iopub.status.idle":"2022-06-14T21:02:25.558047Z","shell.execute_reply.started":"2022-06-14T21:02:25.551689Z","shell.execute_reply":"2022-06-14T21:02:25.557091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_params = {\n        'loss_function' : 'Logloss',\n        'eval_metric' : 'AUC',\n        'learning_rate': 0.1,\n        'num_boost_round': 2500,\n        'early_stopping_rounds': 50,\n        'random_state': 127,\n        'task_type': 'GPU'\n    }","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:02:26.806803Z","iopub.execute_input":"2022-06-14T21:02:26.807369Z","iopub.status.idle":"2022-06-14T21:02:26.811991Z","shell.execute_reply.started":"2022-06-14T21:02:26.807333Z","shell.execute_reply":"2022-06-14T21:02:26.811220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_feats = []\nfi_df_all = []\nwhile True:\n    train_x, valid_x, train_y, valid_y = train_test_split(data[use_cols].to_pandas(), data[TARGET].to_pandas(), test_size=0.33, random_state=42)\n    \n    trn_data = Pool(train_x, label=train_y)\n    val_data = Pool(valid_x ,label=valid_y)\n\n    model = CatBoost(cat_params)\n    model.fit(trn_data,\n            eval_set=val_data,\n            verbose_eval=500,\n            use_best_model=True\n          )\n    pred = model.predict(val_data)\n    auc_score = roc_auc_score(valid_y,pred)\n    print(f'AUC Score : {auc_score}')\n    \n    # time savings\n    if auc_score < 0.75:\n        break\n    else:\n        fi_df = pd.DataFrame()\n        fi_df['importance'] = model.get_feature_importance(Pool(train_x, train_y))\n        fi_df['features'] = use_cols\n        fi_df = fi_df.sort_values(by=\"importance\", ascending=False)\n        fi_df_all.append(fi_df)\n        drop_feats += fi_df['features'].to_list()[:5]\n        use_cols = [c for c in use_cols if c not in drop_feats]\n        print(f'Drop Features : {drop_feats}')","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:02:47.017528Z","iopub.execute_input":"2022-06-14T21:02:47.018234Z","iopub.status.idle":"2022-06-14T21:09:58.326535Z","shell.execute_reply.started":"2022-06-14T21:02:47.018181Z","shell.execute_reply":"2022-06-14T21:09:58.325058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Features","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,15))\nfor i,feat in enumerate(drop_feats):\n    plt.subplot(5,5,i+1)\n    sns.distplot(train[feat].to_pandas(),label='Train')\n    sns.distplot(test_public[feat].to_pandas(),label='public')\n    plt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:10:07.402704Z","iopub.execute_input":"2022-06-14T21:10:07.403081Z","iopub.status.idle":"2022-06-14T21:11:18.544485Z","shell.execute_reply.started":"2022-06-14T21:10:07.403048Z","shell.execute_reply":"2022-06-14T21:11:18.543599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Null Ratio","metadata":{}},{"cell_type":"code","source":"for feat in drop_feats:\n    print(f'========================= {feat} =========================')\n    print('Train Nan Ratio:',train[feat].isnull().sum()/len(train))\n    print('Public Nan Ratio:',test_public[feat].isnull().sum()/len(test_public))","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:11:32.300394Z","iopub.execute_input":"2022-06-14T21:11:32.300746Z","iopub.status.idle":"2022-06-14T21:11:32.341720Z","shell.execute_reply.started":"2022-06-14T21:11:32.300715Z","shell.execute_reply":"2022-06-14T21:11:32.340969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train vs Private","metadata":{}},{"cell_type":"code","source":"data = cudf.concat([train,test_private]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:11:46.627105Z","iopub.execute_input":"2022-06-14T21:11:46.627458Z","iopub.status.idle":"2022-06-14T21:11:46.776131Z","shell.execute_reply.started":"2022-06-14T21:11:46.627429Z","shell.execute_reply":"2022-06-14T21:11:46.775336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET = 'target'\ndrop_cols = ['S_2','month','customer_ID',TARGET]\nuse_cols = [c for c in data.columns if c not in drop_cols]","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:11:47.825389Z","iopub.execute_input":"2022-06-14T21:11:47.826376Z","iopub.status.idle":"2022-06-14T21:11:47.831875Z","shell.execute_reply.started":"2022-06-14T21:11:47.826330Z","shell.execute_reply":"2022-06-14T21:11:47.831135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_params = {\n        'loss_function' : 'Logloss',\n        'eval_metric' : 'AUC',\n        'learning_rate': 0.1,\n        'num_boost_round': 2500,\n        'early_stopping_rounds': 50,\n        'random_state': 127,\n        'task_type': 'GPU'\n    }","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:11:55.636468Z","iopub.execute_input":"2022-06-14T21:11:55.636808Z","iopub.status.idle":"2022-06-14T21:11:55.641187Z","shell.execute_reply.started":"2022-06-14T21:11:55.636779Z","shell.execute_reply":"2022-06-14T21:11:55.640437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_feats = []\nfi_df_all = []\nwhile True:\n    train_x, valid_x, train_y, valid_y = train_test_split(data[use_cols].to_pandas(), data[TARGET].to_pandas(), test_size=0.33, random_state=42)\n    \n    trn_data = Pool(train_x, label=train_y)\n    val_data = Pool(valid_x ,label=valid_y)\n\n    model = CatBoost(cat_params)\n    model.fit(trn_data,\n            eval_set=val_data,\n            verbose_eval=500,\n            use_best_model=True\n          )\n    pred = model.predict(val_data)\n    auc_score = roc_auc_score(valid_y,pred)\n    print(f'AUC Score : {auc_score}')\n    \n    # time savings\n    if auc_score < 0.75:\n        break\n    else:\n        fi_df = pd.DataFrame()\n        fi_df['importance'] = model.get_feature_importance(Pool(train_x, train_y))\n        fi_df['features'] = use_cols\n        fi_df = fi_df.sort_values(by=\"importance\", ascending=False)\n        fi_df_all.append(fi_df)\n        drop_feats += fi_df['features'].to_list()[:5]\n        use_cols = [c for c in use_cols if c not in drop_feats]\n        print(f'Drop Features : {drop_feats}')","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:12:04.859316Z","iopub.execute_input":"2022-06-14T21:12:04.859674Z","iopub.status.idle":"2022-06-14T21:15:52.969627Z","shell.execute_reply.started":"2022-06-14T21:12:04.859644Z","shell.execute_reply":"2022-06-14T21:15:52.968757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Features","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,15))\nfor i,feat in enumerate(drop_feats):\n    plt.subplot(3,5,i+1)\n    sns.distplot(train[feat].to_pandas(),label='Train')\n    sns.distplot(test_private[feat].to_pandas(),label='private')\n    plt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:15:52.971180Z","iopub.execute_input":"2022-06-14T21:15:52.971627Z","iopub.status.idle":"2022-06-14T21:16:38.490832Z","shell.execute_reply.started":"2022-06-14T21:15:52.971587Z","shell.execute_reply":"2022-06-14T21:16:38.490109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Null Ratio","metadata":{}},{"cell_type":"code","source":"for feat in drop_feats:\n    print(f'========================= {feat} =========================')\n    print('Train Nan Ratio:',train[feat].isnull().sum()/len(train))\n    print('Public Nan Ratio:',test_private[feat].isnull().sum()/len(test_private))","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:16:38.492093Z","iopub.execute_input":"2022-06-14T21:16:38.492598Z","iopub.status.idle":"2022-06-14T21:16:38.518381Z","shell.execute_reply.started":"2022-06-14T21:16:38.492557Z","shell.execute_reply":"2022-06-14T21:16:38.517537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Public vs Private","metadata":{}},{"cell_type":"code","source":"public_drop_feats = ['B_29', 'R_1', 'D_59', 'S_11', 'S_15', 'S_9', 'S_24', 'D_121', 'S_27', 'S_22', 'D_45', 'R_27', 'D_62', 'S_13', 'D_91', 'D_39', 'D_42', 'D_77', 'B_8', 'D_142', 'P_4', 'B_17', 'P_3', 'D_120', 'S_17']\nprivate_drop_feats = ['R_1', 'D_59', 'S_11', 'S_9', 'S_27', 'D_121', 'R_27', 'S_15', 'S_22', 'S_24', 'D_39', 'D_62', 'D_45', 'B_17', 'D_60']\nonly_private_drop_feats = [f for f in private_drop_feats if f not in public_drop_feats]\nonly_private_drop_feats","metadata":{"execution":{"iopub.status.busy":"2022-06-14T21:17:02.166931Z","iopub.execute_input":"2022-06-14T21:17:02.167590Z","iopub.status.idle":"2022-06-14T21:17:02.174862Z","shell.execute_reply.started":"2022-06-14T21:17:02.167554Z","shell.execute_reply":"2022-06-14T21:17:02.174130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}