{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T17:50:13.182015Z","iopub.execute_input":"2022-08-13T17:50:13.182492Z","iopub.status.idle":"2022-08-13T17:50:13.188106Z","shell.execute_reply.started":"2022-08-13T17:50:13.182457Z","shell.execute_reply":"2022-08-13T17:50:13.186929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('seaborn')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.247374Z","iopub.execute_input":"2022-08-13T17:50:13.247790Z","iopub.status.idle":"2022-08-13T17:50:13.254807Z","shell.execute_reply.started":"2022-08-13T17:50:13.247759Z","shell.execute_reply":"2022-08-13T17:50:13.253001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References\n* https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense/notebook\n* https://www.kaggle.com/sdysch/tps-aug-2022/","metadata":{}},{"cell_type":"markdown","source":"# Reading in the data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ndf_test = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.316358Z","iopub.execute_input":"2022-08-13T17:50:13.316857Z","iopub.status.idle":"2022-08-13T17:50:13.609130Z","shell.execute_reply.started":"2022-08-13T17:50:13.316801Z","shell.execute_reply":"2022-08-13T17:50:13.608071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['id'], axis='columns')\ndf_test = df_test.drop(['id'], axis='columns')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.610743Z","iopub.execute_input":"2022-08-13T17:50:13.611615Z","iopub.status.idle":"2022-08-13T17:50:13.622275Z","shell.execute_reply.started":"2022-08-13T17:50:13.611577Z","shell.execute_reply":"2022-08-13T17:50:13.620659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGBoost model","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(['failure', 'product_code'], axis='columns')\ny = df_train['failure']\n\ncat_cols = [v for v in X.columns if X[v].dtype in ['object', 'int' ]]\nnumerical_cols = [v for v in X.columns if v not in cat_cols]\n\nint_cols = [\n    # 'attribute_0',\n    # 'attribute_1',\n    'attribute_2',\n    'attribute_3',\n]\n\nprint(cat_cols)\nprint(numerical_cols)\n\nxgb_weight = df_train[df_train['failure'] == 1].shape[0] / df_train[df_train['failure'] == 0].shape[0]\nprint(xgb_weight)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.623912Z","iopub.execute_input":"2022-08-13T17:50:13.624582Z","iopub.status.idle":"2022-08-13T17:50:13.651695Z","shell.execute_reply.started":"2022-08-13T17:50:13.624545Z","shell.execute_reply":"2022-08-13T17:50:13.650072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Variable transformations","metadata":{}},{"cell_type":"code","source":"cols = [f'measurement_{i}' for i in range(18)]\nfor var in cols:\n    df_train[var] /= df_train['loading']\n    df_test[var] /= df_test['loading']\n# df_train = df_train.drop(['loading'], axis='columns')\n# df_test = df_test.drop(['loading'], axis='columns')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.654942Z","iopub.execute_input":"2022-08-13T17:50:13.655958Z","iopub.status.idle":"2022-08-13T17:50:13.684389Z","shell.execute_reply.started":"2022-08-13T17:50:13.655877Z","shell.execute_reply":"2022-08-13T17:50:13.683105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameter tuning","metadata":{}},{"cell_type":"markdown","source":"## Objective function","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_validate\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder, LabelBinarizer, StandardScaler\nfrom sklearn.compose import ColumnTransformer\nfrom xgboost import XGBClassifier\nimport optuna\n\ndef objective(trial):\n    \n    # hyperparameter space\n    # XGBoost\n    \n    n_estimators = trial.suggest_int('n_estimators', 10, 200, 10)\n    eta = trial.suggest_float('eta', 0.05, 0.3, step=0.05)\n    max_depth = trial.suggest_int('max_depth', 2, 5)\n    gamma = trial.suggest_int('gamma', 0, 5)\n    \n    # colsample_bytree = trial.suggest_float('colsample_bytree', 0, 1, step=0.1)\n    # colsample_bylevel = trial.suggest_float('colsample_bylevel', 0, 1, step=0.1)\n    # colsample_bynode = trial.suggest_float('colsample_bynode', 0, 1, step=0.1)\n    # subsample = trial.suggest_float('subsample', 0.5, 1.0, step=0.25)\n\n    \n    # other\n    # add_indicator = trial.suggest_categorical('add_indicator', [True, False])\n    add_indicator = True\n    # class_weight = trial.suggest_categorical('class_weight', [xgb_weight, 1])\n    \n    # pipeline setup\n    \n    # impute missing values\n    imputer = Pipeline(\n        [\n            ('knn_imputer', KNNImputer(add_indicator=add_indicator))\n        ],\n    )\n    \n    # OneHotEncode categorical features\n    category_transformer = Pipeline(\n        [\n            ('OneHotEncoder', OneHotEncoder(handle_unknown='ignore', sparse=False)),\n        ],\n    )\n\n    # total preprocessing\n    preproc = ColumnTransformer(\n        transformers = [\n            ('KNNImputer', imputer, numerical_cols),\n            ('OneHotEncoder', category_transformer, cat_cols),\n        ], remainder='passthrough'\n    )\n\n    # FIXME class weight balanced?\n    model = XGBClassifier(\n        objective='binary:logistic',\n        max_depth=max_depth,\n        n_estimators=n_estimators,\n        eta=eta,\n        gamma=gamma,\n        # colsample_bytree=colsample_bytree,\n        # colsample_bylevel=colsample_bylevel,\n        # colsample_bynode=colsample_bynode,\n        # subsample=subsample,\n        # scale_pos_weight=class_weight,\n    )\n\n    # final model\n    _pipe = Pipeline(\n        [\n            ('Preprocessing', preproc),\n            ('XGBClassifer', model),\n        ]\n    )\n    \n    # scoring\n    scoring = ['roc_auc']\n    scores = list()\n\n    # CV\n    groups = df_train['product_code']\n    cv = GroupKFold(len(groups.unique()))\n    cv_results = cross_validate(_pipe, X, y, cv=cv, scoring=scoring, groups=groups)\n    \n    return np.mean(cv_results['test_roc_auc'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.686193Z","iopub.execute_input":"2022-08-13T17:50:13.686581Z","iopub.status.idle":"2022-08-13T17:50:13.701050Z","shell.execute_reply.started":"2022-08-13T17:50:13.686548Z","shell.execute_reply":"2022-08-13T17:50:13.699667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run trials","metadata":{}},{"cell_type":"code","source":"\"\"\"# optimise\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=100)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.702576Z","iopub.execute_input":"2022-08-13T17:50:13.703015Z","iopub.status.idle":"2022-08-13T17:50:13.720176Z","shell.execute_reply.started":"2022-08-13T17:50:13.702978Z","shell.execute_reply":"2022-08-13T17:50:13.718958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"print(f'Best trial: {study.best_trial}')\nprint(f'Best value: {study.best_value}')\nprint(f'Best parameters: {study.best_params}')\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.778893Z","iopub.execute_input":"2022-08-13T17:50:13.780349Z","iopub.status.idle":"2022-08-13T17:50:13.787554Z","shell.execute_reply.started":"2022-08-13T17:50:13.780298Z","shell.execute_reply":"2022-08-13T17:50:13.786050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Best model","metadata":{}},{"cell_type":"code","source":"# pipeline setup\n\n# impute missing values\nimputer = Pipeline(\n    [\n        ('knn_imputer', KNNImputer(add_indicator=True))\n    ],\n)\n\n# OneHotEncode categorical features\ncategory_transformer = Pipeline(\n    [\n        ('OneHotEncoder', OneHotEncoder(handle_unknown='ignore', sparse=False)),\n    ],\n)\n\n# total preprocessing\npreproc = ColumnTransformer(\n    transformers = [\n        ('KNNImputer', imputer, numerical_cols),\n        ('OneHotEncoder', category_transformer, cat_cols),\n    ], remainder='passthrough'\n)\n\nmodel = XGBClassifier(\n    objective='binary:logistic',\n    max_depth=2,\n    n_estimators=70,\n    eta=0.25,\n    gamma=5,\n)\n\n# final model\npipe = Pipeline(\n    [\n        ('Preprocessing', preproc),\n        ('XGBClassifer', model),\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.880807Z","iopub.execute_input":"2022-08-13T17:50:13.881284Z","iopub.status.idle":"2022-08-13T17:50:13.892396Z","shell.execute_reply.started":"2022-08-13T17:50:13.881245Z","shell.execute_reply":"2022-08-13T17:50:13.890579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"pipe.fit(X, y)\npred = pipe.predict_proba(df_test)\npred","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:50:13.947430Z","iopub.execute_input":"2022-08-13T17:50:13.947910Z","iopub.status.idle":"2022-08-13T17:51:27.928904Z","shell.execute_reply.started":"2022-08-13T17:50:13.947864Z","shell.execute_reply":"2022-08-13T17:51:27.927694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv')\nsubmission['failure'] = pred[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:51:27.931415Z","iopub.execute_input":"2022-08-13T17:51:27.932292Z","iopub.status.idle":"2022-08-13T17:51:27.950137Z","shell.execute_reply.started":"2022-08-13T17:51:27.932237Z","shell.execute_reply":"2022-08-13T17:51:27.948476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:51:27.951464Z","iopub.execute_input":"2022-08-13T17:51:27.951820Z","iopub.status.idle":"2022-08-13T17:51:28.001614Z","shell.execute_reply.started":"2022-08-13T17:51:27.951783Z","shell.execute_reply":"2022-08-13T17:51:28.000167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:51:28.003942Z","iopub.execute_input":"2022-08-13T17:51:28.004334Z","iopub.status.idle":"2022-08-13T17:51:28.021772Z","shell.execute_reply.started":"2022-08-13T17:51:28.004300Z","shell.execute_reply":"2022-08-13T17:51:28.020365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature importances","metadata":{}},{"cell_type":"code","source":"#print(model)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T17:51:28.023703Z","iopub.execute_input":"2022-08-13T17:51:28.024117Z","iopub.status.idle":"2022-08-13T17:51:28.028576Z","shell.execute_reply.started":"2022-08-13T17:51:28.024085Z","shell.execute_reply":"2022-08-13T17:51:28.027347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}