{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-14T13:10:35.756581Z","iopub.execute_input":"2023-03-14T13:10:35.757040Z","iopub.status.idle":"2023-03-14T13:10:35.788760Z","shell.execute_reply.started":"2023-03-14T13:10:35.756938Z","shell.execute_reply":"2023-03-14T13:10:35.788019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thank you CHRIS DEOTTE for a great job - https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676\n","metadata":{}},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom sklearn.model_selection import KFold, GroupKFold, StratifiedKFold\nfrom sklearn.metrics import f1_score\nfrom catboost import CatBoostClassifier\nimport matplotlib.pyplot as plt\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nimport lightgbm as lgb\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.decomposition import PCA\nfrom sklearn.svm import SVC\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.model_selection import RepeatedKFold","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:10:37.199268Z","iopub.execute_input":"2023-03-14T13:10:37.199806Z","iopub.status.idle":"2023-03-14T13:10:39.023142Z","shell.execute_reply.started":"2023-03-14T13:10:37.199774Z","shell.execute_reply":"2023-03-14T13:10:39.022195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\nprint( train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:10:39.024685Z","iopub.execute_input":"2023-03-14T13:10:39.024993Z","iopub.status.idle":"2023-03-14T13:11:38.276369Z","shell.execute_reply.started":"2023-03-14T13:10:39.024951Z","shell.execute_reply":"2023-03-14T13:11:38.275427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:11:38.277786Z","iopub.execute_input":"2023-03-14T13:11:38.278102Z","iopub.status.idle":"2023-03-14T13:11:38.875992Z","shell.execute_reply.started":"2023-03-14T13:11:38.278074Z","shell.execute_reply":"2023-03-14T13:11:38.875252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:11:38.877907Z","iopub.execute_input":"2023-03-14T13:11:38.878397Z","iopub.status.idle":"2023-03-14T13:11:38.883478Z","shell.execute_reply.started":"2023-03-14T13:11:38.878368Z","shell.execute_reply":"2023-03-14T13:11:38.882640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)   \n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sem')\n        tmp.name = tmp.name + '_sem'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('median')\n        tmp.name = tmp.name + '_median'\n        dfs.append(tmp) \n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('var')\n        tmp.name = tmp.name + '_var'\n        dfs.append(tmp)    \n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:23:09.359987Z","iopub.execute_input":"2023-03-14T13:23:09.360347Z","iopub.status.idle":"2023-03-14T13:23:09.372684Z","shell.execute_reply.started":"2023-03-14T13:23:09.360317Z","shell.execute_reply":"2023-03-14T13:23:09.371877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = feature_engineer(train)\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:23:09.949033Z","iopub.execute_input":"2023-03-14T13:23:09.949574Z","iopub.status.idle":"2023-03-14T13:24:51.039077Z","shell.execute_reply.started":"2023-03-14T13:23:09.949542Z","shell.execute_reply":"2023-03-14T13:24:51.038152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:24:51.040662Z","iopub.execute_input":"2023-03-14T13:24:51.041070Z","iopub.status.idle":"2023-03-14T13:24:51.047177Z","shell.execute_reply.started":"2023-03-14T13:24:51.041041Z","shell.execute_reply":"2023-03-14T13:24:51.046519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:24:51.048368Z","iopub.execute_input":"2023-03-14T13:24:51.048895Z","iopub.status.idle":"2023-03-14T13:24:51.062209Z","shell.execute_reply.started":"2023-03-14T13:24:51.048866Z","shell.execute_reply":"2023-03-14T13:24:51.061382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=20)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels_xgb = {}\nmodels_cat = {}\nmodels_lgb = {}\nmodels_hgb = {}\n\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf_xgb =  XGBClassifier(**xgb_params)\n        clf_xgb.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf_xgb.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models_xgb[f'{grp}_{t}'] = clf_xgb\n        \n        # TRAIN MODEL\n        clf_cat = CatBoostClassifier(\n            iterations = 1000,\n            depth = 4,\n            learning_rate = 0.05,\n            loss_function = \"Logloss\",\n            random_seed = 0,\n            metric_period = 1,\n            subsample = 0.8,\n            colsample_bylevel = 0.4,\n            verbose = 0\n        )       \n        clf_cat.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                    eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                    verbose=0)\n        print(f'{t}({clf_cat.best_iteration_}), ',end='') \n        \n        models_cat[f'{grp}_{t}'] = clf_cat\n        oof.loc[valid_users, t-1] = clf_cat.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n        lgb_params = {\n            'objective' : 'binary',\n            'metric' : 'auc',\n            'learning_rate': 0.002,\n            'max_depth': 6,\n            'n_estimators': 1000,\n            'verbose': -1\n        }\n        clf_lgb =  lgb.LGBMClassifier(**lgb_params)\n        clf_lgb.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                    eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ], verbose = -1)\n        #print(f'{t}({clf_lgb.best_ntree_limit}), ',end='')\n        models_lgb[f'{grp}_{t}'] = clf_lgb\n        \n        hgb_param = {'learning_rate': 0.009918121506128766,\n                     'max_depth': 27,\n                     'max_bins': 243,\n                     'max_leaf_nodes': 46}\n        \n        clf_hgb =  HistGradientBoostingClassifier()\n        clf_hgb.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        models_hgb[f'{grp}_{t}'] = clf_hgb\n                       \n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:24:51.064328Z","iopub.execute_input":"2023-03-14T13:24:51.064925Z","iopub.status.idle":"2023-03-14T13:56:00.261162Z","shell.execute_reply.started":"2023-03-14T13:24:51.064896Z","shell.execute_reply":"2023-03-14T13:56:00.260229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute CV Score","metadata":{}},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:00.262835Z","iopub.execute_input":"2023-03-14T13:56:00.263158Z","iopub.status.idle":"2023-03-14T13:56:00.332136Z","shell.execute_reply.started":"2023-03-14T13:56:00.263130Z","shell.execute_reply":"2023-03-14T13:56:00.331267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:00.333541Z","iopub.execute_input":"2023-03-14T13:56:00.333835Z","iopub.status.idle":"2023-03-14T13:56:03.214473Z","shell.execute_reply.started":"2023-03-14T13:56:00.333808Z","shell.execute_reply":"2023-03-14T13:56:03.213504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:03.215706Z","iopub.execute_input":"2023-03-14T13:56:03.216011Z","iopub.status.idle":"2023-03-14T13:56:03.494102Z","shell.execute_reply.started":"2023-03-14T13:56:03.215983Z","shell.execute_reply":"2023-03-14T13:56:03.493230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:03.495259Z","iopub.execute_input":"2023-03-14T13:56:03.495558Z","iopub.status.idle":"2023-03-14T13:56:03.656106Z","shell.execute_reply.started":"2023-03-14T13:56:03.495530Z","shell.execute_reply":"2023-03-14T13:56:03.655356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test Data","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel train, targets, df, oof, true\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:03.657563Z","iopub.execute_input":"2023-03-14T13:56:03.658136Z","iopub.status.idle":"2023-03-14T13:56:03.802440Z","shell.execute_reply.started":"2023-03-14T13:56:03.658105Z","shell.execute_reply":"2023-03-14T13:56:03.801575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\ndef norm(preds):\n    return (preds - np.min(preds)) / (np.max(preds) - np.min(preds))\n\nfor (sample_submission, test) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf1 = models_xgb[f'{grp}_{t}']\n        clf2 = models_cat[f'{grp}_{t}']\n        clf3 = models_lgb[f'{grp}_{t}']\n        clf4 = models_hgb[f'{grp}_{t}']\n        \n        p1 = clf1.predict_proba(df[FEATURES].astype('float32'))[:,1] \n        p2 = clf2.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        p3 = clf3.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        p4 = clf4.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        \n        p = (p1+p2+p3+p4)/4                        #clf.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold)\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:03.805286Z","iopub.execute_input":"2023-03-14T13:56:03.805584Z","iopub.status.idle":"2023-03-14T13:56:05.590997Z","shell.execute_reply.started":"2023-03-14T13:56:03.805557Z","shell.execute_reply":"2023-03-14T13:56:05.589940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:05.592149Z","iopub.execute_input":"2023-03-14T13:56:05.592939Z","iopub.status.idle":"2023-03-14T13:56:05.605278Z","shell.execute_reply.started":"2023-03-14T13:56:05.592908Z","shell.execute_reply":"2023-03-14T13:56:05.604176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"execution":{"iopub.status.busy":"2023-03-14T13:56:05.606642Z","iopub.execute_input":"2023-03-14T13:56:05.606933Z","iopub.status.idle":"2023-03-14T13:56:05.611323Z","shell.execute_reply.started":"2023-03-14T13:56:05.606907Z","shell.execute_reply":"2023-03-14T13:56:05.610641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}