{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Adversarial Validation","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n%matplotlib inline\nplt.rcParams[\"figure.figsize\"] = (14, 8)\nsns.set_theme(context=\"notebook\", style=\"whitegrid\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:32.737198Z","iopub.execute_input":"2022-08-04T08:01:32.737723Z","iopub.status.idle":"2022-08-04T08:01:33.367753Z","shell.execute_reply.started":"2022-08-04T08:01:32.737608Z","shell.execute_reply":"2022-08-04T08:01:33.366492Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"# file paths\nDATA_DIR = Path(\"../input/tabular-playground-series-aug-2022\")\n\n# data\nTRAIN_DATA = DATA_DIR / \"train.csv\"\n\nTEST_DATA = DATA_DIR / \"test.csv\"\n\n# columns in the data\nINDEX_COL = \"id\"\n\nTARGET_COL = \"failure\"\n\nCONTINUOUS_FEATURES_REGEX = r\"^measure\\w+|^load\\w+\"\n\nDISCRETE_FEATURES_REGEX = r\"^attr\\w+|\\w+code$\"\n\n# random state\nRANDOM_SEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:33.369927Z","iopub.execute_input":"2022-08-04T08:01:33.370304Z","iopub.status.idle":"2022-08-04T08:01:33.376803Z","shell.execute_reply.started":"2022-08-04T08:01:33.370271Z","shell.execute_reply":"2022-08-04T08:01:33.375546Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(TRAIN_DATA, index_col=INDEX_COL)\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:33.378492Z","iopub.execute_input":"2022-08-04T08:01:33.379478Z","iopub.status.idle":"2022-08-04T08:01:33.549160Z","shell.execute_reply.started":"2022-08-04T08:01:33.379437Z","shell.execute_reply":"2022-08-04T08:01:33.547649Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(TEST_DATA, index_col=INDEX_COL)\ntest_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:33.553346Z","iopub.execute_input":"2022-08-04T08:01:33.554268Z","iopub.status.idle":"2022-08-04T08:01:33.658863Z","shell.execute_reply.started":"2022-08-04T08:01:33.554213Z","shell.execute_reply":"2022-08-04T08:01:33.657629Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn import set_config\nfrom sklearn.compose import make_column_transformer, make_column_selector\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom xgboost import XGBClassifier\n\nset_config(display=\"diagram\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:33.660469Z","iopub.execute_input":"2022-08-04T08:01:33.660873Z","iopub.status.idle":"2022-08-04T08:01:33.753443Z","shell.execute_reply.started":"2022-08-04T08:01:33.660812Z","shell.execute_reply":"2022-08-04T08:01:33.752183Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_trans = make_column_transformer(\n    (\n        OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-999),\n        make_column_selector(pattern=DISCRETE_FEATURES_REGEX)\n    ),\n    remainder=\"passthrough\",\n    n_jobs=-1\n)\n\npipe = Pipeline([\n    (\"transformer\", col_trans),\n    (\"model\", XGBClassifier(random_state=RANDOM_SEED))\n])\npipe","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:33.755570Z","iopub.execute_input":"2022-08-04T08:01:33.756467Z","iopub.status.idle":"2022-08-04T08:01:33.795512Z","shell.execute_reply.started":"2022-08-04T08:01:33.756415Z","shell.execute_reply":"2022-08-04T08:01:33.794122Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"from sklearn.base import clone\nfrom sklearn.model_selection import cross_val_score, train_test_split\nimport shap","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:33.797176Z","iopub.execute_input":"2022-08-04T08:01:33.798287Z","iopub.status.idle":"2022-08-04T08:01:34.820341Z","shell.execute_reply.started":"2022-08-04T08:01:33.798229Z","shell.execute_reply":"2022-08-04T08:01:34.818904Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# utility functions\ndef create_dataset(train, other, target_col=\"is_test\", drop_cols=None):\n    try:\n        train = train.drop(TARGET_COL, axis=1)\n    except KeyError as e:\n        print(e)\n\n    train[target_col] = 0\n    other[target_col] = 1\n    df = pd.concat([train, other]).sample(\n        frac=1, random_state=RANDOM_SEED\n    )\n    print(df[target_col].value_counts())\n    \n    # drop columns\n    if drop_cols is not None:\n        df.drop(drop_cols, axis=1, inplace=True)\n    \n    # separate features from target\n    y = df[target_col]\n    X = df.drop(target_col, axis=1)\n    return X, y\n\ndef get_shap_values(pipe, X, y, trans=\"transformer\"):\n    X_train, X_val, y_train, y_val = train_test_split(\n        X, y, random_state=RANDOM_SEED\n    )\n    pipe = clone(pipe)\n    pipe.fit(X_train, y_train)\n    transformer = pipe.named_steps[trans]\n    X_val_trans = transformer.transform(X_val)\n\n    explainer = shap.TreeExplainer(pipe.named_steps[\"model\"])\n    shap_values = explainer.shap_values(X_val_trans)\n    return pipe, shap_values, X_val_trans","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:34.825768Z","iopub.execute_input":"2022-08-04T08:01:34.826893Z","iopub.status.idle":"2022-08-04T08:01:34.840169Z","shell.execute_reply.started":"2022-08-04T08:01:34.826818Z","shell.execute_reply":"2022-08-04T08:01:34.839138Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### With all features","metadata":{}},{"cell_type":"code","source":"X, y = create_dataset(train_df, test_df)\ncv_results = cross_val_score(pipe, X, y, scoring=\"roc_auc\")\ncv_results = pd.DataFrame(cv_results, columns=[\"AUC\"])\ncv_results.describe().loc[[\"mean\", \"std\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:34.841394Z","iopub.execute_input":"2022-08-04T08:01:34.842600Z","iopub.status.idle":"2022-08-04T08:01:48.717285Z","shell.execute_reply.started":"2022-08-04T08:01:34.842550Z","shell.execute_reply":"2022-08-04T08:01:48.715891Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe_, shap_values, X_val_trans = get_shap_values(pipe, X, y)\nshap.summary_plot(\n    shap_values,\n    features=X_val_trans,\n    feature_names=pipe_.named_steps[\"transformer\"].feature_names_in_,\n    class_names=pipe_.named_steps[\"model\"].classes_,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:48.719220Z","iopub.execute_input":"2022-08-04T08:01:48.719623Z","iopub.status.idle":"2022-08-04T08:01:54.539297Z","shell.execute_reply.started":"2022-08-04T08:01:48.719585Z","shell.execute_reply":"2022-08-04T08:01:54.538201Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Without product code","metadata":{}},{"cell_type":"code","source":"X, y = create_dataset(train_df, test_df, drop_cols=\"product_code\")\ncv_results = cross_val_score(pipe, X, y, scoring=\"roc_auc\")\ncv_results = pd.DataFrame(cv_results, columns=[\"AUC\"])\ncv_results.describe().loc[[\"mean\", \"std\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:01:54.540935Z","iopub.execute_input":"2022-08-04T08:01:54.542148Z","iopub.status.idle":"2022-08-04T08:02:12.523972Z","shell.execute_reply.started":"2022-08-04T08:01:54.542101Z","shell.execute_reply":"2022-08-04T08:02:12.522979Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe_, shap_values, X_val_trans = get_shap_values(pipe, X, y)\nshap.summary_plot(\n    shap_values,\n    features=X_val_trans,\n    feature_names=pipe_.named_steps[\"transformer\"].feature_names_in_,\n    class_names=pipe_.named_steps[\"model\"].classes_,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:02:12.525293Z","iopub.execute_input":"2022-08-04T08:02:12.526440Z","iopub.status.idle":"2022-08-04T08:02:19.568262Z","shell.execute_reply.started":"2022-08-04T08:02:12.526399Z","shell.execute_reply":"2022-08-04T08:02:19.566977Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Without product code and attributes","metadata":{}},{"cell_type":"code","source":"drop_cols = [\n    \"product_code\", \"attribute_0\", \"attribute_1\", \"attribute_2\", \"attribute_3\"\n]\nX, y = create_dataset(train_df, test_df, drop_cols=drop_cols)\ncv_results = cross_val_score(pipe, X, y, scoring=\"roc_auc\")\ncv_results = pd.DataFrame(cv_results, columns=[\"AUC\"])\ncv_results.describe().loc[[\"mean\", \"std\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:02:19.570703Z","iopub.execute_input":"2022-08-04T08:02:19.571641Z","iopub.status.idle":"2022-08-04T08:03:04.607409Z","shell.execute_reply.started":"2022-08-04T08:02:19.571604Z","shell.execute_reply":"2022-08-04T08:03:04.606188Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe_, shap_values, X_val_trans = get_shap_values(pipe, X, y)\nshap.summary_plot(\n    shap_values,\n    features=X_val_trans,\n    feature_names=pipe_.named_steps[\"transformer\"].feature_names_in_,\n    class_names=pipe_.named_steps[\"model\"].classes_,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:03:04.608949Z","iopub.execute_input":"2022-08-04T08:03:04.609304Z","iopub.status.idle":"2022-08-04T08:03:21.107045Z","shell.execute_reply.started":"2022-08-04T08:03:04.609272Z","shell.execute_reply":"2022-08-04T08:03:21.105989Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}