{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## LGBM Regressor Ensemble Pipeline\nThis notebook broadly follows pipeline introduced in Chris Deotte's baseline [Random Forest](https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664) and XGBoost notebooks. There are 2-3 major modifications above it as mentioned in the respective places:\n\n**(1)** Features\n**(2)** Model\n**(3)** Ensembling\n\n**Please leave an upvote if you found it helpful!**","metadata":{}},{"cell_type":"markdown","source":"<div style=\"text-align: center;\">Major LB scores so far</div>\n\n| Algo                                | LB    |\n|-------------------------------------|-------|\n| All Zeros                           | 0.226 |\n| All Ones                            | 0.414 |\n| Rounding Question Means at 0.5      | 0.587 |\n| Question Means w/ optimum threshold | 0.648 |\n| Random Forest Baseline              | 0.664 |\n| XGBoost Baseline                    | 0.676 |\n","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nfrom tqdm import tqdm\nfrom collections import Counter\nimport itertools\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import f1_score\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom lightgbm.sklearn import LGBMRegressor","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:03:13.886754Z","iopub.execute_input":"2023-02-15T04:03:13.887316Z","iopub.status.idle":"2023-02-15T04:03:15.675673Z","shell.execute_reply.started":"2023-02-15T04:03:13.887183Z","shell.execute_reply":"2023-02-15T04:03:15.674598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntrain_labels = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:03:15.677632Z","iopub.execute_input":"2023-02-15T04:03:15.678250Z","iopub.status.idle":"2023-02-15T04:04:36.659285Z","shell.execute_reply.started":"2023-02-15T04:03:15.678190Z","shell.execute_reply":"2023-02-15T04:04:36.658303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:04:36.660875Z","iopub.execute_input":"2023-02-15T04:04:36.661467Z","iopub.status.idle":"2023-02-15T04:04:37.247294Z","shell.execute_reply.started":"2023-02-15T04:04:36.661422Z","shell.execute_reply":"2023-02-15T04:04:37.246141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\nWe create basic aggregate features. Try creating more features to boost CV and LB! The idea for EVENTS feature is from [here][1]\n\n[1]: https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\n\nWe add 4 extra features over the baseline here - max time, min time, time duration of the level group, and average time of events.","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:04:37.249708Z","iopub.execute_input":"2023-02-15T04:04:37.250186Z","iopub.status.idle":"2023-02-15T04:04:37.256397Z","shell.execute_reply.started":"2023-02-15T04:04:37.250150Z","shell.execute_reply":"2023-02-15T04:04:37.254973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n    \n    tmp = train.groupby(['session_id', 'level_group'])['elapsed_time'].agg(['max', 'min', 'count'])\n    tmp['time'] = tmp['max'] - tmp['min']\n    tmp['avg_time'] = tmp['time']/tmp['count']\n    dfs.append(tmp)\n    \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    \n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:04:37.258190Z","iopub.execute_input":"2023-02-15T04:04:37.258647Z","iopub.status.idle":"2023-02-15T04:04:37.272980Z","shell.execute_reply.started":"2023-02-15T04:04:37.258610Z","shell.execute_reply":"2023-02-15T04:04:37.271665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:04:37.274991Z","iopub.execute_input":"2023-02-15T04:04:37.275424Z","iopub.status.idle":"2023-02-15T04:05:46.067494Z","shell.execute_reply.started":"2023-02-15T04:04:37.275387Z","shell.execute_reply":"2023-02-15T04:05:46.066413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:05:51.675850Z","iopub.execute_input":"2023-02-15T04:05:51.676916Z","iopub.status.idle":"2023-02-15T04:05:51.853012Z","shell.execute_reply.started":"2023-02-15T04:05:51.676875Z","shell.execute_reply":"2023-02-15T04:05:51.852027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training\n\nWe'll be using a Light GBM Regressor here!","metadata":{}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:05:54.777365Z","iopub.execute_input":"2023-02-15T04:05:54.778146Z","iopub.status.idle":"2023-02-15T04:05:54.787997Z","shell.execute_reply.started":"2023-02-15T04:05:54.778104Z","shell.execute_reply":"2023-02-15T04:05:54.786828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_FOLDS = 10\n\ngkf = GroupKFold(n_splits=N_FOLDS)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH N GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        model = LGBMRegressor(learning_rate=0.027, \\\n                    num_leaves=15, \\\n                    n_estimators=200, \\\n                    min_child_samples=20, \\\n                    boosting_type='gbdt',\n                    subsample_for_bin=1000,\n                    max_depth=-1,\n                    colsample_bytree=0.8)\n        model.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{i}_{grp}_{t}'] = model\n        oof.loc[valid_users, t-1] = model.predict(valid_x[FEATURES])\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:05:56.758366Z","iopub.execute_input":"2023-02-15T04:05:56.759181Z","iopub.status.idle":"2023-02-15T04:07:41.142022Z","shell.execute_reply.started":"2023-02-15T04:05:56.759140Z","shell.execute_reply":"2023-02-15T04:07:41.140961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute CV Score\nWe need to convert prediction probabilities into `1s` and `0s`. The competition metric is F1 Score which is the harmonic mean of precision and recall. Let's find the optimal threshold for `p > threshold` when to predict `1` and when to predict `0` to maximize F1 Score.","metadata":{}},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:09:35.762909Z","iopub.execute_input":"2023-02-15T04:09:35.763432Z","iopub.status.idle":"2023-02-15T04:09:35.858108Z","shell.execute_reply.started":"2023-02-15T04:09:35.763391Z","shell.execute_reply":"2023-02-15T04:09:35.856862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:09:36.039686Z","iopub.execute_input":"2023-02-15T04:09:36.040119Z","iopub.status.idle":"2023-02-15T04:09:40.354950Z","shell.execute_reply.started":"2023-02-15T04:09:36.040085Z","shell.execute_reply":"2023-02-15T04:09:40.353592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.4f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:09:40.356713Z","iopub.execute_input":"2023-02-15T04:09:40.357187Z","iopub.status.idle":"2023-02-15T04:09:40.636451Z","shell.execute_reply.started":"2023-02-15T04:09:40.357151Z","shell.execute_reply":"2023-02-15T04:09:40.635458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel targets, df, oof, true\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:09:49.110844Z","iopub.execute_input":"2023-02-15T04:09:49.111379Z","iopub.status.idle":"2023-02-15T04:09:49.265631Z","shell.execute_reply.started":"2023-02-15T04:09:49.111338Z","shell.execute_reply":"2023-02-15T04:09:49.264077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We ensemble our models from all the different folds!**","metadata":{}},{"cell_type":"code","source":"def predict(qid, grp, test):\n    val = 0\n    for fold in range(N_FOLDS):\n        val += models[f'{fold}_{grp}_{qid}'].predict(test[FEATURES])[0]\n    return val > best_threshold*N_FOLDS","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:10:04.203005Z","iopub.execute_input":"2023-02-15T04:10:04.203573Z","iopub.status.idle":"2023-02-15T04:10:04.211548Z","shell.execute_reply.started":"2023-02-15T04:10:04.203529Z","shell.execute_reply":"2023-02-15T04:10:04.209916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (sample_submission, test) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    sample_submission['qid'] = sample_submission['session_id'].apply(lambda x: x.split(\"_\")[1][1:]).astype(int)\n    sample_submission['correct'] = sample_submission['qid'].apply(lambda x: predict(x, grp, df)).astype(int)\n    del sample_submission['qid']\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:10:04.844030Z","iopub.execute_input":"2023-02-15T04:10:04.845355Z","iopub.status.idle":"2023-02-15T04:10:06.829899Z","shell.execute_reply.started":"2023-02-15T04:10:04.845301Z","shell.execute_reply":"2023-02-15T04:10:06.828949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"submission.csv\")\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-15T04:10:06.831525Z","iopub.execute_input":"2023-02-15T04:10:06.832193Z","iopub.status.idle":"2023-02-15T04:10:06.847254Z","shell.execute_reply.started":"2023-02-15T04:10:06.832154Z","shell.execute_reply":"2023-02-15T04:10:06.846145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}