{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!/usr/bin/env python\n# coding: utf-8\n\n# ## LGBM Regressor Ensemble Pipeline\n# This notebook broadly follows pipeline introduced in Chris Deotte's baseline [Random Forest](https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664) and XGBoost notebooks. There are 2-3 major modifications above it as mentioned in the respective places:\n# \n# **(1)** Features\n# **(2)** Model\n# **(3)** Ensembling\n# \n# **Please leave an upvote if you found it helpful!**\n\n# <div style=\"text-align: center;\">Major LB scores so far</div>\n# \n# | Algo                                | LB    |\n# |-------------------------------------|-------|\n# | All Zeros                           | 0.226 |\n# | All Ones                            | 0.414 |\n# | Rounding Question Means at 0.5      | 0.587 |\n# | Question Means w/ optimum threshold | 0.648 |\n# | Random Forest Baseline              | 0.664 |\n# | XGBoost Baseline                    | 0.676 |\n# \n\n# In[ ]:\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport gc\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, \\\n    roc_curve, auc\nfrom sklearn.metrics import roc_auc_score\n\nimport pickle\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom lightgbm.sklearn import LGBMRegressor\nimport lightgbm as lgb\nfrom matplotlib import ticker\nimport time\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score\n\nnotebook = True\n\n\ndef reduce_memory_usage(df):\n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024 ** 2\n    print(\"Memory usage became: \", mem_usg, \" MB\")\n\n    return df\n\n\ndtypes = {\n    'elapsed_time': np.int32,\n    'event_name': 'category',\n    'name': 'category',\n    'level': np.uint8,\n    'room_coor_x': np.float32,\n    'room_coor_y': np.float32,\n    'screen_coor_x': np.float32,\n    'screen_coor_y': np.float32,\n    'hover_duration': np.float32,\n    'text': 'category',\n    'fqid': 'category',\n    'room_fqid': 'category',\n    'text_fqid': 'category',\n    'fullscreen': 'category',\n    'hq': 'category',\n    'music': 'category',\n    'level_group': 'category'}\n# if notebook:\n#     train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\n#     targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n# else:\n#     train = pd.read_feather('data123/train.feather')\n#     targets = pd.read_csv('data/train_labels.csv')\ntry:\n    train = pd.read_feather('data/train.feather')\n    targets = pd.read_csv('data/train_labels.csv')\nexcept:\n    train = pd.read_feather('/kaggle/input/data-train2/train.feather')\n    targets = pd.read_csv('/kaggle/input/data-train2/train_labels.csv')\ntrain = reduce_memory_usage(train)\n\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\n\n# In[ ]:\n\n\n# # Feature Engineer\n# We create basic aggregate features. Try creating more features to boost CV and LB! The idea for EVENTS feature is from [here][1]\n# \n# [1]: https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\n# \n# We add 4 extra features over the baseline here - max time, min time, time duration of the level group, and average time of events.\n\n# In[ ]:\n\n\nCATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time', 'level', 'page', 'room_coor_x', 'room_coor_y',\n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click', 'person_click', 'cutscene_click', 'object_click',\n          'map_hover', 'notification_click', 'map_click', 'observation_click',\n          'checkpoint']\n\n\n# In[ ]:\n\n\ndef feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS:\n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS, axis=1)\n\n    tmp = train.groupby(['session_id', 'level_group'])['elapsed_time'].agg(['max', 'min', 'count'])\n    tmp['time'] = tmp['max'] - tmp['min']\n    tmp['avg_time'] = tmp['time'] / tmp['count']\n    dfs.append(tmp)\n\n    df = pd.concat(dfs, axis=1)\n    df = df.fillna(-1)\n\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df\n\n\n# In[ ]:\n\n\n# In[ ]:\n\ndf = feature_engineer(train)\nprint(df.shape)\ndf.head()\ndel train\ngc.collect()\n\n# In[ ]:\n\nFEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES), 'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS), 'users info')\n\n# In[ ]:\n\n\nN_FOLDS = 5\n\ngkf = GroupKFold(n_splits=N_FOLDS)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\nmodels = {}\n\nmax_depth = -1\nseed = 42\nearly_stopping_rounds = 500\nparams = {\n    # 'num_leaves': 2 ** (max_depth - 1),\n    'objective': 'binary',\n    # 'metric': 'rmse',\n    # Feiyang: 11. 把 8 改成了 9\n    'max_depth': max_depth,\n    # 'min_data_in_leaf': 20,\n    'n_estimators': 6000,\n    'learning_rate': 0.027,\n    # 'num_leaves':15,\n    'min_child_samples': 20,\n    # 'colsample_bytree': 0.8,\n    'min_child_samples': 20,\n    'subsample_for_bin': 1000,\n    # 'feature_fraction': 0.6,\n    # 'bagging_fraction': 0.8,\n    # 'bagging_freq': 1,\n    'metric': 'auc',\n    'n_jobs': -1,\n    # 'is_unbalance': True,\n    'seed': seed,\n}\n# COMPUTE CV SCORE WITH N GROUP K FOLD\nfor i, (train_index, val_index) in enumerate(gkf.split(X=df, groups=df.index)):\n\n    print('#' * 25)\n    print('### Fold', i + 1)\n    print('#' * 25)\n\n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1, 19):\n\n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t <= 3:\n            grp = '0-4'\n        elif t <= 13:\n            grp = '5-12'\n        elif t <= 22:\n            grp = '13-22'\n\n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp, FEATURES].astype('float32')\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q == t].set_index('session').loc[train_users, 'correct']\n\n        # VALID DATA\n        valid_x = df.iloc[val_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp, FEATURES].astype('float32')\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q == t].set_index('session').loc[valid_users, 'correct']\n\n        # TRAIN MODEL\n\n        # model = LGBMRegressor(learning_rate=0.027, \\\n        #                       num_leaves=15, \\\n        #                       n_estimators=300, \\\n        #                       min_child_samples=20, \\\n        #                       boosting_type='gbdt',\n        #                       subsample_for_bin=1000,\n        #                       max_depth=-1,\n        #                       colsample_bytree=0.8)\n\n        # model.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        model = lgb.LGBMClassifier(**params)\n        model.fit(train_x, train_y, eval_set=[(train_x, train_y), (valid_x, valid_y)], eval_names=['train', 'val'],\n                  early_stopping_rounds=early_stopping_rounds, verbose=-1)\n        oof.loc[valid_users, t - 1] = model.predict_proba(valid_x)[:, 1]\n        # model = LGBMRegressor(learning_rate=0.027,\n        #                       num_leaves=15,\n        #                       n_estimators=300,\n        #                       min_child_samples=20,\n        #                       boosting_type='gbdt',\n        #                       subsample_for_bin=1000,\n        #                       max_depth=-1,\n        #                       colsample_bytree=0.8,\n        #                       seed=seed)\n\n        # oof.loc[valid_users, t - 1] = model.predict(valid_x)\n\n        val_r2 = roc_auc_score(valid_y, oof.loc[valid_users, t - 1])\n        print(f'fold{i + 1}_q{t}_val_auc_score:{val_r2:.4f}')\n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{i}_{grp}_{t}'] = model\n\n    print()\n\n# # Compute CV Score\n# We need to convert prediction probabilities into `1s` and `0s`. The competition metric is F1 Score which is the harmonic mean of precision and recall. Let's find the optimal threshold for `p > threshold` when to predict `1` and when to predict `0` to maximize F1 Score.\n\n# In[ ]:\n\n\n# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k + 1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values\n\n# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []\nthresholds = []\nbest_score = 0\nbest_threshold = 0\n\nfor threshold in np.arange(0.4, 0.81, 0.01):\n    # print(f'{threshold:.02f}, ', end='')\n    preds = (oof.values.reshape((-1)) > threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\nprint(f'\\nBest F1 Score = {best_score:.4f} at threshold = {best_threshold:.3f}')\n\n# import matplotlib.pyplot as plt\n#\n# # PLOT THRESHOLD VS. F1_SCORE\n# plt.figure(figsize=(20, 5))\n# plt.plot(thresholds, scores, '-o', color='blue')\n# plt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\n# plt.xlabel('Threshold', size=14)\n# plt.ylabel('Validation F1 Score', size=14)\n# plt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.4f} at Best Threshold = {best_threshold:.3}',\n#           size=18)\n# plt.show()\n\n# # Inference\n\n# In[ ]:\n\n\ntry:\n    import jo_wilder\n    env = jo_wilder.make_env()\n    iter_test = env.iter_test()\nexcept:\n    print('no jo_wilder')\n    notebook = False\nif notebook:\n    # CLEAR MEMORY\n    import gc\n\n    del targets, df, oof, true\n    _ = gc.collect()\n\n\n    def predict(qid, grp, test):\n        val = 0\n        for fold in range(N_FOLDS):\n            val += models[f'{fold}_{grp}_{qid}'].predict_proba(test[FEATURES])[:, 1]\n\n        return val / N_FOLDS > best_threshold\n\n\n    for (test, sample_submission) in iter_test:\n        # FEATURE ENGINEER TEST DATA\n        df = feature_engineer(test)\n\n        # INFER TEST DATA\n        grp = test.level_group.values[0]\n        sample_submission['qid'] = sample_submission['session_id'].apply(lambda x: x.split(\"_\")[1][1:]).astype(int)\n        sample_submission['correct'] = sample_submission['qid'].apply(lambda x: predict(x, grp, df)).astype(int)\n        del sample_submission['qid']\n\n        env.predict(sample_submission)\n\n    sub = pd.read_csv(\"submission.csv\")\n    sub.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-19T01:45:42.219042Z","iopub.execute_input":"2023-05-19T01:45:42.219595Z","iopub.status.idle":"2023-05-19T01:54:18.276270Z","shell.execute_reply.started":"2023-05-19T01:45:42.219496Z","shell.execute_reply":"2023-05-19T01:54:18.275245Z"},"trusted":true},"execution_count":null,"outputs":[]}]}