{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\n\ntrain = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\nprint( train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T05:50:22.064264Z","iopub.execute_input":"2023-03-12T05:50:22.064698Z","iopub.status.idle":"2023-03-12T05:51:29.806367Z","shell.execute_reply.started":"2023-03-12T05:50:22.064659Z","shell.execute_reply":"2023-03-12T05:51:29.805368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T05:58:11.094302Z","iopub.execute_input":"2023-03-12T05:58:11.094773Z","iopub.status.idle":"2023-03-12T05:58:11.710645Z","shell.execute_reply.started":"2023-03-12T05:58:11.09473Z","shell.execute_reply":"2023-03-12T05:58:11.709679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CATS is a list of categorical features in the data.\n# NUMS is a list of numerical features in the data.\n# EVENTS is a list of specific events in the game.\n# feature_engineer function takes the training data and \n# does some feature engineering on it. It creates new \n# features by grouping the data by session ID and level\n# group, and then aggregating the data in different ways\n# for each categorical and numerical feature. It also \n# creates new binary features \n# for each event in the EVENTS list, and sums up the \n# elapsed_time feature. The function returns a new DataFrame\n# with the engineered features.\n# df is the DataFrame of the engineered features.\n# FEATURES is a list of feature names to use for training the model.\n# ALL_USERS is a list of all the session IDs in the data.\n# gkf is a GroupKFold object that is used to perform cross-validation\n# on the data.\n# oof is a DataFrame that is used to store the out-of-fold predictions\n# for each question and each user.\n# models is a dictionary that is used to store the trained models\n# for each question and each level group.\n# The for loop performs cross-validation on the data using the\n# GroupKFold object. For each fold, it trains a separate XGBoost\n# model for each question and each level group, and saves the\n# trained models in the models dictionary. It also predicts\n# the out-of-fold probabilities for each question and each \n# user, and stores them in the oof DataFrame.\n# true is a copy of the oof DataFrame, but with the correct\n# labels for each question and each user.\n# The for loop at the end finds the best threshold to use \n# for converting the probabilities in the oof DataFrame into\n# binary predictions. It does this by iterating over a range\n# of possible threshold values and computing the macro F1 \n# score for each threshold. The best threshold is the one that\n# gives the highest F1 score, and is stored in the best_threshold\n# variable.\n\n\nCATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']\n\ndef feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df\n\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()\n\nFEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')\n\ngkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()\n\n# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values\n\n# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This code is plotting the F1 score of the \n# model against different threshold values, and identifying\n# the optimal threshold value that maximizes the F1 score.\n\n# After finding the best threshold, it computes the F1 \n# score for each question separately and prints the results.\n# Finally, it computes the overall F1 score\n# of the model and prints the result.\n\nimport matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()\n\nprint('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T05:45:51.000171Z","iopub.execute_input":"2023-03-12T05:45:51.000646Z","iopub.status.idle":"2023-03-12T05:45:51.467401Z","shell.execute_reply.started":"2023-03-12T05:45:51.000608Z","shell.execute_reply":"2023-03-12T05:45:51.466145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}