{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Simple step by step baseline\n\nThis notebook is a simple baseline with step by step implementation of\n*     1. Data Pre-processing\n*     2. Cross-validation\n*     3. Fitting Model\n*     4. Hyperparameter tuning\n    ","metadata":{}},{"cell_type":"markdown","source":"## Importing necessary libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nfrom sklearn.model_selection import GroupKFold, train_test_split\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, PowerTransformer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score, roc_curve\n\nimport xgboost as xgb\nfrom catboost import CatBoostClassifier\nimport optuna\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T13:32:59.856915Z","iopub.execute_input":"2022-08-12T13:32:59.857407Z","iopub.status.idle":"2022-08-12T13:32:59.868843Z","shell.execute_reply.started":"2022-08-12T13:32:59.857365Z","shell.execute_reply":"2022-08-12T13:32:59.867852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading raw files","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ndf_test = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:33:01.806813Z","iopub.execute_input":"2022-08-12T13:33:01.807166Z","iopub.status.idle":"2022-08-12T13:33:02.054384Z","shell.execute_reply.started":"2022-08-12T13:33:01.807133Z","shell.execute_reply":"2022-08-12T13:33:02.053480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical column names in a list\ncat_cols = ['attribute_2', 'attribute_3', 'attribute_0', 'attribute_1']\n# Numerical column names in a list\nnum_cols = [col for col in df_train.columns if (col.startswith('measurement'))|( col == 'loading')]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:33:03.445095Z","iopub.execute_input":"2022-08-12T13:33:03.445475Z","iopub.status.idle":"2022-08-12T13:33:03.451823Z","shell.execute_reply.started":"2022-08-12T13:33:03.445442Z","shell.execute_reply":"2022-08-12T13:33:03.449962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transform the input data\n\n**Note** creating a function to take as input, raw train and test dataframes along with the categorical and numerical columns names.\n\nThis function will transform the train, test dataframes and output the prepared dataframe to be fed to model.\n\nOne hot encoding, imputation and scaling/standardization methods are to be fit on training data only and used for transforming traning and testing data\n","metadata":{}},{"cell_type":"code","source":"def prepare_df(train, test, categorical_cols, numeric_cols):\n    \n    #one hot encode categorical features\n    categories = []\n    for c in categorical_cols:\n        cats = list(train[c].unique())\n        categories.append(sorted(list(cats)))\n     \n    # since there are unknown categories in test set we are adding \"handle unknown = 'ignore'\"\n    ohe = OneHotEncoder(categories = categories,\n                        sparse = False,\n                        handle_unknown = 'ignore')\n    ohe.fit(train[categorical_cols])\n    \n    # get transformed feature names\n    cat_out = ohe.get_feature_names_out()\n    train[cat_out] = ohe.transform(train[categorical_cols])\n    test[cat_out] = ohe.transform(test[categorical_cols])\n    \n    #droping original categorical features as we have OHE transformed\n    train.drop(columns = categorical_cols, axis = 1, inplace = True)\n    test.drop(columns = categorical_cols, axis = 1, inplace = True)\n    \n    # finding columns that have missing elements\n    missing = df_train.isna().sum()\n    missing_cols = [col for col in missing.index if missing[col] > 0]\n    \n    # Performing imputation\n    imputer = KNNImputer(n_neighbors=3)\n    imputer.fit(train[missing_cols])\n    train[missing_cols] = imputer.transform(train[missing_cols])\n    test[missing_cols] = imputer.transform(test[missing_cols])\n    \n    # getting cols to be scaled\n    cols_scale = [col for col in train.columns if col not in ('id', 'failure', 'product_code')]\n\n    # scaling features so that all features are in same scale\n    scaler = StandardScaler()\n    scaler.fit(train[cols_scale])\n    train[cols_scale] = scaler.transform(train[cols_scale])\n    test[cols_scale] = scaler.transform(test[cols_scale])\n    \n    return train, test","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:33:06.035499Z","iopub.execute_input":"2022-08-12T13:33:06.036644Z","iopub.status.idle":"2022-08-12T13:33:06.047479Z","shell.execute_reply.started":"2022-08-12T13:33:06.036598Z","shell.execute_reply":"2022-08-12T13:33:06.046415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# applying the function on the raw train and test data\ntrain, test = prepare_df(df_train, df_test, cat_cols, num_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:33:07.027984Z","iopub.execute_input":"2022-08-12T13:33:07.028968Z","iopub.status.idle":"2022-08-12T13:33:53.082914Z","shell.execute_reply.started":"2022-08-12T13:33:07.028919Z","shell.execute_reply":"2022-08-12T13:33:53.081934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross-validation function\n\nThis cross validation function takes the prepared train and test data creates a groupKFold cross validation scheme, Fits the input models and outputs the AUC scores, and list of test scores created by the 5 CV models.\n\nThis Group KFold is creating 5 folds for the 5 product_codes in the training data, so our input train set will be broken such that 4 product will be train set and 1 product will be validation set.\n","metadata":{}},{"cell_type":"code","source":"def get_scores(model_name, model, Xtrain, Ytrain, test):\n    \n    # input lists for appending\n    auc_scores = []\n    test_preds_list = []\n    \n    # creating groupKfold object\n    kf= GroupKFold(n_splits = 5)\n    \n    # running loop for every fold based on product code\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(Xtrain, Ytrain, Xtrain['product_code'])):\n        \n        # creating training and validation sets based on index values\n        X_train = Xtrain.iloc[idx_tr]\n        X_valid = Xtrain.iloc[idx_va]\n        X_train.drop('product_code', axis =1, inplace = True)\n        X_valid.drop('product_code', axis =1, inplace = True)\n        X_test = test.copy()\n        X_test.drop('product_code', axis =1, inplace = True)\n        Y_train = Ytrain.iloc[idx_tr]\n        Y_valid = Ytrain.iloc[idx_va]\n        \n        # fitting the passed model\n        model.fit(X_train, Y_train)\n        \n        # predict the probabilities and take the second column and validation score\n        va_preds = model.predict_proba(X_valid)[:,1]\n        score = roc_auc_score(Y_valid, va_preds)\n        auc_scores.append(score)\n        \n        # similarly for test set\n        test_preds = model.predict_proba(X_test)[:,1]\n        test_preds_list.append(test_preds)\n        \n        print(f\"{model_name} fold: {fold} AUC: {score}\")\n        \n    mean_score = np.mean(auc_scores)\n    \n    return mean_score, auc_scores, test_preds_list","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:34:03.840102Z","iopub.execute_input":"2022-08-12T13:34:03.840800Z","iopub.status.idle":"2022-08-12T13:34:03.849699Z","shell.execute_reply.started":"2022-08-12T13:34:03.840761Z","shell.execute_reply":"2022-08-12T13:34:03.848362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create model with base parameters\nmodel = LogisticRegression(penalty='l1', C=0.01,solver='liblinear', random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:35:08.727231Z","iopub.execute_input":"2022-08-12T13:35:08.727603Z","iopub.status.idle":"2022-08-12T13:35:08.732774Z","shell.execute_reply.started":"2022-08-12T13:35:08.727571Z","shell.execute_reply":"2022-08-12T13:35:08.731780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  fit the model and predict the probabilities","metadata":{}},{"cell_type":"code","source":"\n# we call the finction created above which crossvalidates and provides CV scores and test scores\n# below score is the mean AUC score of CV,list of AUC scores, list of predicted probabilities\nscore, aucs, test_preds_list = get_scores(model_name = 'Logit',\n                                                    model = model,\n                                                    Xtrain = train.drop('failure', axis = 1),\n                                                    Ytrain = train['failure'],\n                                                    test = test)\n\n# since there was cross validation and 5 models predicted 5 times the scores we are averaging the predictions\nfinal = sum(test_preds_list)/len(test_preds_list)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:35:11.140622Z","iopub.execute_input":"2022-08-12T13:35:11.141560Z","iopub.status.idle":"2022-08-12T13:35:13.417682Z","shell.execute_reply.started":"2022-08-12T13:35:11.141523Z","shell.execute_reply":"2022-08-12T13:35:13.416672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit and keep on improving the solution","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\nsubmission['failure'] = final\nsubmission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:04:08.354573Z","iopub.status.idle":"2022-08-11T20:04:08.355059Z","shell.execute_reply.started":"2022-08-11T20:04:08.354798Z","shell.execute_reply":"2022-08-11T20:04:08.354833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameter tuning using optuna\n\nWe can now take a more complex model to fit and find the hyperparameters to tune.\n\nWe take XGBoost to train and optimize\n\noptuna needs an objective function which will fit the model along with parameters we pass. this funciton should return the score that we want to optimize in this case ROC AUC.","metadata":{}},{"cell_type":"code","source":"\ndef objective(trial,data = train.drop('product_code', axis = 1), target = train['failure']):\n    \n    # first Train test split he dataset\n    train_x, test_x, train_y, test_y = train_test_split(data, target, test_size = 0.25, random_state = 42)\n    \n    # dictionary of parameters we wish to optimize\n    # there are 3 main choices trail.suggest-categorical/int/float/loguniform which is the \n    # distribution to choose from the numbers we pass\n    param = {\n        \n        'tree_method':'gpu_hist',  #turn on GPU for this parameter else comment out\n        'lambda': trial.suggest_loguniform('lambda', 1e-3, 10.0),\n        'alpha': trial.suggest_loguniform('alpha', 1e-3, 10.0),\n        'colsample_bytree': trial.suggest_categorical('colsample_bytree', [0.3,0.4,0.5,0.6,0.7,0.8,0.9, 1.0]),\n        'subsample': trial.suggest_categorical('subsample', [0.4,0.5,0.6,0.7,0.8,1.0]),\n        'learning_rate': trial.suggest_categorical('learning_rate', [0.008,0.01,0.012,0.014,0.016,0.018, 0.02]),\n        'n_estimators': 10000,\n        'max_depth': trial.suggest_categorical('max_depth', [5,7,9,11,13,15,17]),\n        'random_state': trial.suggest_categorical('random_state', [2020]),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 300),\n        'early_stopping_rounds': 100\n    }\n    # fit and predict return the score\n    model = xgb.XGBClassifier(**param)\n    model.fit(train_x, train_y, eval_set = [(test_x, test_y)],   verbose = False)\n    preds = model.predict_proba(test_x)[:,1]\n    score = roc_auc_score(test_y, preds)\n    return score","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:36:27.005423Z","iopub.execute_input":"2022-08-12T13:36:27.006043Z","iopub.status.idle":"2022-08-12T13:36:27.019142Z","shell.execute_reply.started":"2022-08-12T13:36:27.006006Z","shell.execute_reply":"2022-08-12T13:36:27.018197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a study and specify if we wish to maximize the score or minimize the score\nstudy = optuna.create_study(direction = 'maximize')\nstudy.optimize(objective, n_trials = 30)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:36:38.999567Z","iopub.execute_input":"2022-08-12T13:36:39.000131Z","iopub.status.idle":"2022-08-12T13:37:46.589583Z","shell.execute_reply.started":"2022-08-12T13:36:39.000093Z","shell.execute_reply":"2022-08-12T13:37:46.588358Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_best_params = study.best_params\nprint('Number of finished trials:', len(study.trials))\nprint('Best trial:', study.best_trial.params)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:38:15.935228Z","iopub.execute_input":"2022-08-12T13:38:15.935978Z","iopub.status.idle":"2022-08-12T13:38:15.947364Z","shell.execute_reply.started":"2022-08-12T13:38:15.935941Z","shell.execute_reply":"2022-08-12T13:38:15.946172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using the best parameters create a model and check\nxgb_model = xgb.XGBClassifier(**study.best_params)\nscore, aucs, test_preds_list = get_scores(model_name = 'XGBclassifier',\n                                          model = xgb_model,\n                                          Xtrain = train.drop('failure', axis=1),\n                                          Ytrain = train['failure'],\n                                          test = test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:38:23.340982Z","iopub.execute_input":"2022-08-12T13:38:23.341364Z","iopub.status.idle":"2022-08-12T13:39:13.692538Z","shell.execute_reply.started":"2022-08-12T13:38:23.341308Z","shell.execute_reply":"2022-08-12T13:39:13.691731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Keep experimenting using the functions we just created","metadata":{}}]}