{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Vowpal Wabbit Baseline - LB 0.65\n\nThis notebook is just a copy of Chris Deotte's series of baseline notebooks (but here using an online learner):\nhttps://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676\n\nIn this notebook we present a Vowal Wabbit baseline. We train GroupKFold models for each of the 18 questions. Our CV score is 0.664. We infer test using one of our KFold models. We can improve our CV and LB by engineering more features for our random forest and/or trying different models (like other ML models and/or RNN and/or Transformer). Also we can improve our LB by using more KFold models OR training one model using all data (and the hyperparameters that we found from our KFold cross validation).","metadata":{"papermill":{"duration":0.005932,"end_time":"2023-02-07T00:59:58.147501","exception":false,"start_time":"2023-02-07T00:59:58.141569","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score","metadata":{"papermill":{"duration":1.027875,"end_time":"2023-02-07T00:59:59.180261","exception":false,"start_time":"2023-02-07T00:59:58.152386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:15:16.670311Z","iopub.execute_input":"2023-02-08T02:15:16.671515Z","iopub.status.idle":"2023-02-08T02:15:18.228258Z","shell.execute_reply.started":"2023-02-08T02:15:16.671388Z","shell.execute_reply":"2023-02-08T02:15:18.227034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from vowpalwabbit.sklearn_vw import VWClassifier","metadata":{"execution":{"iopub.status.busy":"2023-02-12T09:32:39.530502Z","iopub.execute_input":"2023-02-12T09:32:39.531810Z","iopub.status.idle":"2023-02-12T09:32:40.237193Z","shell.execute_reply.started":"2023-02-12T09:32:39.531685Z","shell.execute_reply":"2023-02-12T09:32:40.236214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\nprint( train.shape )\ntrain.head()","metadata":{"papermill":{"duration":59.284316,"end_time":"2023-02-07T01:00:58.478743","exception":false,"start_time":"2023-02-07T00:59:59.194427","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:15:18.232624Z","iopub.execute_input":"2023-02-08T02:15:18.233023Z","iopub.status.idle":"2023-02-08T02:16:30.528172Z","shell.execute_reply.started":"2023-02-08T02:15:18.232987Z","shell.execute_reply":"2023-02-08T02:16:30.527228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"papermill":{"duration":0.598155,"end_time":"2023-02-07T01:00:59.082015","exception":false,"start_time":"2023-02-07T01:00:58.48386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:16:30.529716Z","iopub.execute_input":"2023-02-08T02:16:30.530419Z","iopub.status.idle":"2023-02-08T02:16:31.178221Z","shell.execute_reply.started":"2023-02-08T02:16:30.53037Z","shell.execute_reply":"2023-02-08T02:16:31.177003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\nWe create basic aggregate features. Try creating more features to boost CV and LB!","metadata":{"papermill":{"duration":0.005196,"end_time":"2023-02-07T01:00:59.092865","exception":false,"start_time":"2023-02-07T01:00:59.087669","status":"completed"},"tags":[]}},{"cell_type":"code","source":"CATS = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"papermill":{"duration":0.014685,"end_time":"2023-02-07T01:00:59.112856","exception":false,"start_time":"2023-02-07T01:00:59.098171","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:16:31.18085Z","iopub.execute_input":"2023-02-08T02:16:31.181499Z","iopub.status.idle":"2023-02-08T02:16:31.187166Z","shell.execute_reply.started":"2023-02-08T02:16:31.181453Z","shell.execute_reply":"2023-02-08T02:16:31.18574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"papermill":{"duration":0.017716,"end_time":"2023-02-07T01:00:59.136021","exception":false,"start_time":"2023-02-07T01:00:59.118305","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:16:31.18878Z","iopub.execute_input":"2023-02-08T02:16:31.189177Z","iopub.status.idle":"2023-02-08T02:16:31.201175Z","shell.execute_reply.started":"2023-02-08T02:16:31.189144Z","shell.execute_reply":"2023-02-08T02:16:31.199728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":34.516494,"end_time":"2023-02-07T01:01:33.658043","exception":false,"start_time":"2023-02-07T01:00:59.141549","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:16:31.202903Z","iopub.execute_input":"2023-02-08T02:16:31.20349Z","iopub.status.idle":"2023-02-08T02:17:13.054521Z","shell.execute_reply.started":"2023-02-08T02:16:31.20344Z","shell.execute_reply":"2023-02-08T02:17:13.053627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Vowpal Wabbit Model\nWe train one model for each of 18 questions. Furthermore, we use data from `level_groups = '0-4'` to train model for questions 1-3, and `level groups '5-12'` to train questions 4 thru 13 and `level groups '13-22'` to train questions 14 thru 18. Because this is the data we get (to predict corresponding questions) from Kaggle's inference API during test inference. We can improve our model by saving a user's previous data from earlier `level_groups` and using that to predict future `level_groups`.","metadata":{"papermill":{"duration":0.00565,"end_time":"2023-02-07T01:01:33.669525","exception":false,"start_time":"2023-02-07T01:01:33.663875","status":"completed"},"tags":[]}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"papermill":{"duration":0.014699,"end_time":"2023-02-07T01:01:33.689953","exception":false,"start_time":"2023-02-07T01:01:33.675254","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:17:13.055716Z","iopub.execute_input":"2023-02-08T02:17:13.056627Z","iopub.status.idle":"2023-02-08T02:17:13.065595Z","shell.execute_reply.started":"2023-02-08T02:17:13.056593Z","shell.execute_reply":"2023-02-08T02:17:13.06449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        clf = VWClassifier()\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"papermill":{"duration":69.877213,"end_time":"2023-02-07T01:02:43.57299","exception":false,"start_time":"2023-02-07T01:01:33.695777","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:17:13.067182Z","iopub.execute_input":"2023-02-08T02:17:13.067632Z","iopub.status.idle":"2023-02-08T02:22:53.017091Z","shell.execute_reply.started":"2023-02-08T02:17:13.0676Z","shell.execute_reply":"2023-02-08T02:22:53.016235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute CV Score\nWe need to convert prediction probabilities into `1s` and `0s`. The competition metric is F1 Score which is the harmonic mean of precision and recall. Let's find the optimal threshold for `p > threshold` when to predict `1` and when to predict `0` to maximize F1 Score.","metadata":{"papermill":{"duration":0.011241,"end_time":"2023-02-07T01:02:43.59638","exception":false,"start_time":"2023-02-07T01:02:43.585139","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-02-08T02:22:56.9952Z","iopub.execute_input":"2023-02-08T02:22:56.99561Z","iopub.status.idle":"2023-02-08T02:22:57.081005Z","shell.execute_reply.started":"2023-02-08T02:22:56.995576Z","shell.execute_reply":"2023-02-08T02:22:57.079834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-02-08T02:27:57.904775Z","iopub.execute_input":"2023-02-08T02:27:57.906012Z","iopub.status.idle":"2023-02-08T02:28:01.965512Z","shell.execute_reply.started":"2023-02-08T02:27:57.90594Z","shell.execute_reply":"2023-02-08T02:28:01.964284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-08T02:31:11.850382Z","iopub.execute_input":"2023-02-08T02:31:11.850818Z","iopub.status.idle":"2023-02-08T02:31:12.060813Z","shell.execute_reply.started":"2023-02-08T02:31:11.850783Z","shell.execute_reply":"2023-02-08T02:31:12.0595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"papermill":{"duration":0.771134,"end_time":"2023-02-07T01:02:44.378465","exception":false,"start_time":"2023-02-07T01:02:43.607331","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-08T02:40:33.915732Z","iopub.execute_input":"2023-02-08T02:40:33.916151Z","iopub.status.idle":"2023-02-08T02:40:34.121926Z","shell.execute_reply.started":"2023-02-08T02:40:33.916116Z","shell.execute_reply":"2023-02-08T02:40:34.121034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test Data","metadata":{"papermill":{"duration":0.011075,"end_time":"2023-02-07T01:02:44.400918","exception":false,"start_time":"2023-02-07T01:02:44.389843","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"papermill":{"duration":0.052132,"end_time":"2023-02-07T01:02:44.464739","exception":false,"start_time":"2023-02-07T01:02:44.412607","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-07T18:36:06.872139Z","iopub.execute_input":"2023-02-07T18:36:06.872474Z","iopub.status.idle":"2023-02-07T18:36:06.884827Z","shell.execute_reply.started":"2023-02-07T18:36:06.872444Z","shell.execute_reply":"2023-02-07T18:36:06.883419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (sample_submission, test) in iter_test:\n    \n    df = feature_engineer(test)\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold)\n    \n    env.predict(sample_submission)","metadata":{"papermill":{"duration":1.002014,"end_time":"2023-02-07T01:02:45.47927","exception":false,"start_time":"2023-02-07T01:02:44.477256","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-07T18:36:10.797568Z","iopub.execute_input":"2023-02-07T18:36:10.797903Z","iopub.status.idle":"2023-02-07T18:36:11.780219Z","shell.execute_reply.started":"2023-02-07T18:36:10.797874Z","shell.execute_reply":"2023-02-07T18:36:11.779548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.011427,"end_time":"2023-02-07T01:02:45.502331","exception":false,"start_time":"2023-02-07T01:02:45.490904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":0.027432,"end_time":"2023-02-07T01:02:45.541022","exception":false,"start_time":"2023-02-07T01:02:45.51359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-07T18:36:14.275251Z","iopub.execute_input":"2023-02-07T18:36:14.275952Z","iopub.status.idle":"2023-02-07T18:36:14.287906Z","shell.execute_reply.started":"2023-02-07T18:36:14.275914Z","shell.execute_reply":"2023-02-07T18:36:14.28663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"papermill":{"duration":0.020233,"end_time":"2023-02-07T01:02:45.57314","exception":false,"start_time":"2023-02-07T01:02:45.552907","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-07T18:36:16.710608Z","iopub.execute_input":"2023-02-07T18:36:16.711012Z","iopub.status.idle":"2023-02-07T18:36:16.716707Z","shell.execute_reply.started":"2023-02-07T18:36:16.710982Z","shell.execute_reply":"2023-02-07T18:36:16.71548Z"},"trusted":true},"execution_count":null,"outputs":[]}]}