{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# XGBoost Baseline \nA great idea from Chris: https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680?scriptVersionId=123110383\n\nThanks to him and all the participants who share their ideas.","metadata":{"papermill":{"duration":0.008113,"end_time":"2023-03-23T09:03:37.158114","exception":false,"start_time":"2023-03-23T09:03:37.150001","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold ,GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score","metadata":{"papermill":{"duration":1.409126,"end_time":"2023-03-23T09:03:38.573510","exception":false,"start_time":"2023-03-23T09:03:37.164384","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:31:20.057813Z","iopub.execute_input":"2023-04-29T06:31:20.058605Z","iopub.status.idle":"2023-04-29T06:31:21.486981Z","shell.execute_reply.started":"2023-04-29T06:31:20.058562Z","shell.execute_reply":"2023-04-29T06:31:21.485998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels\n\nInstead of working on the total train size is 4.7GB, we shrink it into 10 pieces and the feature engineer will read each piece in order.","metadata":{"papermill":{"duration":0.005857,"end_time":"2023-03-23T09:03:38.586274","exception":false,"start_time":"2023-03-23T09:03:38.580417","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')\nPIECES = 10\nCHUNK = int( np.ceil(len(tmp)/PIECES) )\n\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k*CHUNK\n    b = (k+1)*CHUNK\n    if b>len(tmp): b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)\n    \nprint(f'We do this step to avoid memory error')","metadata":{"_kg_hide-input":true,"papermill":{"duration":89.693111,"end_time":"2023-03-23T09:05:08.285527","exception":false,"start_time":"2023-03-23T09:03:38.592416","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:31:21.492872Z","iopub.execute_input":"2023-04-29T06:31:21.495344Z","iopub.status.idle":"2023-04-29T06:32:45.071561Z","shell.execute_reply.started":"2023-04-29T06:31:21.495296Z","shell.execute_reply":"2023-04-29T06:32:45.070647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[0])\nprint('Train size of first piece:', train.shape )\ntrain.head()","metadata":{"papermill":{"duration":7.746988,"end_time":"2023-03-23T09:05:16.038546","exception":false,"start_time":"2023-03-23T09:05:08.291558","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:32:45.073111Z","iopub.execute_input":"2023-04-29T06:32:45.073777Z","iopub.status.idle":"2023-04-29T06:32:52.493606Z","shell.execute_reply.started":"2023-04-29T06:32:45.073741Z","shell.execute_reply":"2023-04-29T06:32:52.492742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"papermill":{"duration":1.215836,"end_time":"2023-03-23T09:05:17.261026","exception":false,"start_time":"2023-03-23T09:05:16.045190","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:32:52.499047Z","iopub.execute_input":"2023-04-29T06:32:52.501546Z","iopub.status.idle":"2023-04-29T06:32:53.673712Z","shell.execute_reply.started":"2023-04-29T06:32:52.501495Z","shell.execute_reply":"2023-04-29T06:32:53.672791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\nWe create basic aggregate features. We think event is important factor for prediction.\n\n[1]: https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data","metadata":{"papermill":{"duration":0.006823,"end_time":"2023-03-23T09:05:17.274574","exception":false,"start_time":"2023-03-23T09:05:17.267751","status":"completed"},"tags":[]}},{"cell_type":"code","source":"CATS = ['event_name','name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# reference: https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"papermill":{"duration":0.01626,"end_time":"2023-03-23T09:05:17.297560","exception":false,"start_time":"2023-03-23T09:05:17.281300","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:32:53.675044Z","iopub.execute_input":"2023-04-29T06:32:53.675633Z","iopub.status.idle":"2023-04-29T06:32:53.681343Z","shell.execute_reply.started":"2023-04-29T06:32:53.675596Z","shell.execute_reply":"2023-04-29T06:32:53.680075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"papermill":{"duration":0.021204,"end_time":"2023-03-23T09:05:17.325758","exception":false,"start_time":"2023-03-23T09:05:17.304554","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:32:53.682706Z","iopub.execute_input":"2023-04-29T06:32:53.683283Z","iopub.status.idle":"2023-04-29T06:32:53.696097Z","shell.execute_reply.started":"2023-04-29T06:32:53.683225Z","shell.execute_reply":"2023-04-29T06:32:53.694806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr = feature_engineer(train)\nprint(df_tr.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T06:32:53.697558Z","iopub.execute_input":"2023-04-29T06:32:53.698178Z","iopub.status.idle":"2023-04-29T06:33:06.491848Z","shell.execute_reply.started":"2023-04-29T06:32:53.698140Z","shell.execute_reply":"2023-04-29T06:33:06.490919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# PROCESS TRAIN DATA IN PIECES\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineer(train)\n    all_pieces.append(df)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train; gc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"papermill":{"duration":357.928294,"end_time":"2023-03-23T09:11:15.261138","exception":false,"start_time":"2023-03-23T09:05:17.332844","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:33:06.493179Z","iopub.execute_input":"2023-04-29T06:33:06.493715Z","iopub.status.idle":"2023-04-29T06:38:39.655142Z","shell.execute_reply.started":"2023-04-29T06:33:06.493682Z","shell.execute_reply":"2023-04-29T06:38:39.654052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train XGBoost Model\nFor each of the 18 questions, we train one model. In addition, we use data from `level_groups = '0-4'`. to train the model for questions 1-3, `level groups '5-12'` to train questions 4-13, and `level groups '13-22'` to train questions 14-18. Because this is the data we acquire from Kaggle's inference API during test inference (to predict corresponding questions). We may improve our model by preserving a user's historical data from previous `level_groups` and utilizing it to forecast future `level_groups`.","metadata":{"papermill":{"duration":0.008136,"end_time":"2023-03-23T09:11:15.277778","exception":false,"start_time":"2023-03-23T09:11:15.269642","status":"completed"},"tags":[]}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"papermill":{"duration":0.024944,"end_time":"2023-03-23T09:11:15.310990","exception":false,"start_time":"2023-03-23T09:11:15.286046","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:38:39.656660Z","iopub.execute_input":"2023-04-29T06:38:39.657224Z","iopub.status.idle":"2023-04-29T06:38:39.673040Z","shell.execute_reply.started":"2023-04-29T06:38:39.657190Z","shell.execute_reply":"2023-04-29T06:38:39.671321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameters Tuning\nWe fine tune the hyperparameters to get the best results.","metadata":{}},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 5,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.8}\n    \n    #'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"papermill":{"duration":159.391009,"end_time":"2023-03-23T09:13:54.710693","exception":false,"start_time":"2023-03-23T09:11:15.319684","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:38:39.677535Z","iopub.execute_input":"2023-04-29T06:38:39.677970Z","iopub.status.idle":"2023-04-29T06:41:16.190798Z","shell.execute_reply.started":"2023-04-29T06:38:39.677928Z","shell.execute_reply":"2023-04-29T06:41:16.189183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute CV Score\nPrediction probabilities must be converted into '1s' and '0s'. The F1 Score, which is the harmonic mean of precision and recall, is the competition metric. To maximize F1 Score, identify the appropriate threshold for 'p > threshold' when to forecast '1' and when to predict '0'.","metadata":{"papermill":{"duration":0.013586,"end_time":"2023-03-23T09:13:54.738201","exception":false,"start_time":"2023-03-23T09:13:54.724615","status":"completed"},"tags":[]}},{"cell_type":"code","source":"true = oof.copy()\nfor k in range(18):\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"papermill":{"duration":0.155933,"end_time":"2023-03-23T09:13:54.907460","exception":false,"start_time":"2023-03-23T09:13:54.751527","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:16.199006Z","iopub.execute_input":"2023-04-29T06:41:16.199610Z","iopub.status.idle":"2023-04-29T06:41:16.338945Z","shell.execute_reply.started":"2023-04-29T06:41:16.199543Z","shell.execute_reply":"2023-04-29T06:41:16.337160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold\n\n\nprint('\\tbest threshold = ',best_threshold)","metadata":{"papermill":{"duration":8.355672,"end_time":"2023-03-23T09:14:03.276543","exception":false,"start_time":"2023-03-23T09:13:54.920871","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:16.340888Z","iopub.execute_input":"2023-04-29T06:41:16.341580Z","iopub.status.idle":"2023-04-29T06:41:23.027716Z","shell.execute_reply.started":"2023-04-29T06:41:16.341540Z","shell.execute_reply":"2023-04-29T06:41:23.026338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"papermill":{"duration":0.30484,"end_time":"2023-03-23T09:14:03.596812","exception":false,"start_time":"2023-03-23T09:14:03.291972","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:23.029602Z","iopub.execute_input":"2023-04-29T06:41:23.029989Z","iopub.status.idle":"2023-04-29T06:41:23.416669Z","shell.execute_reply.started":"2023-04-29T06:41:23.029954Z","shell.execute_reply":"2023-04-29T06:41:23.415782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# As a result \nWe will compute an F1 score for each question and an overall F1 score for all data.","metadata":{}},{"cell_type":"code","source":"print('Applying the best threshold for each question...')\nfor k in range(18):   \n    \n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"papermill":{"duration":0.421191,"end_time":"2023-03-23T09:14:04.034758","exception":false,"start_time":"2023-03-23T09:14:03.613567","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:23.417705Z","iopub.execute_input":"2023-04-29T06:41:23.418029Z","iopub.status.idle":"2023-04-29T06:41:23.744979Z","shell.execute_reply.started":"2023-04-29T06:41:23.417998Z","shell.execute_reply":"2023-04-29T06:41:23.743831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test Data","metadata":{"papermill":{"duration":0.015937,"end_time":"2023-03-23T09:14:04.067487","exception":false,"start_time":"2023-03-23T09:14:04.051550","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel targets, df, oof, true\n_ = gc.collect()","metadata":{"papermill":{"duration":0.187308,"end_time":"2023-03-23T09:14:04.271106","exception":false,"start_time":"2023-03-23T09:14:04.083798","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:23.746381Z","iopub.execute_input":"2023-04-29T06:41:23.746990Z","iopub.status.idle":"2023-04-29T06:41:23.921397Z","shell.execute_reply.started":"2023-04-29T06:41:23.746951Z","shell.execute_reply":"2023-04-29T06:41:23.920006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"papermill":{"duration":0.170596,"end_time":"2023-03-23T09:14:04.457992","exception":false,"start_time":"2023-03-23T09:14:04.287396","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:23.923358Z","iopub.execute_input":"2023-04-29T06:41:23.923755Z","iopub.status.idle":"2023-04-29T06:41:24.864161Z","shell.execute_reply.started":"2023-04-29T06:41:23.923700Z","shell.execute_reply":"2023-04-29T06:41:24.863051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.015671,"end_time":"2023-03-23T09:14:04.489819","exception":false,"start_time":"2023-03-23T09:14:04.474148","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":0.033707,"end_time":"2023-03-23T09:14:04.540409","exception":false,"start_time":"2023-03-23T09:14:04.506702","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:24.865577Z","iopub.execute_input":"2023-04-29T06:41:24.866215Z","iopub.status.idle":"2023-04-29T06:41:24.879529Z","shell.execute_reply.started":"2023-04-29T06:41:24.866173Z","shell.execute_reply":"2023-04-29T06:41:24.878612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"papermill":{"duration":0.026661,"end_time":"2023-03-23T09:14:04.583690","exception":false,"start_time":"2023-03-23T09:14:04.557029","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-29T06:41:24.881316Z","iopub.execute_input":"2023-04-29T06:41:24.881973Z","iopub.status.idle":"2023-04-29T06:41:24.887838Z","shell.execute_reply.started":"2023-04-29T06:41:24.881935Z","shell.execute_reply":"2023-04-29T06:41:24.886916Z"},"trusted":true},"execution_count":null,"outputs":[]}]}