{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Basic Imports\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Common Imports\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.metrics import f1_score\n\n# LGBM Model\nimport lightgbm as lgb\n\n# XGBoost baseline\nfrom xgboost import XGBClassifier\n\n# Random Forest baseline\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-01T21:17:23.875065Z","iopub.execute_input":"2023-03-01T21:17:23.875580Z","iopub.status.idle":"2023-03-01T21:17:26.613390Z","shell.execute_reply.started":"2023-03-01T21:17:23.875474Z","shell.execute_reply":"2023-03-01T21:17:26.612202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reduce Memory Usage\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:17:26.615684Z","iopub.execute_input":"2023-03-01T21:17:26.616052Z","iopub.status.idle":"2023-03-01T21:17:26.645377Z","shell.execute_reply.started":"2023-03-01T21:17:26.616017Z","shell.execute_reply":"2023-03-01T21:17:26.643280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:17:26.651370Z","iopub.execute_input":"2023-03-01T21:17:26.651833Z","iopub.status.idle":"2023-03-01T21:18:17.661877Z","shell.execute_reply.started":"2023-03-01T21:17:26.651789Z","shell.execute_reply":"2023-03-01T21:18:17.660909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = reduce_memory_usage(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:17.663367Z","iopub.execute_input":"2023-03-01T21:18:17.663803Z","iopub.status.idle":"2023-03-01T21:18:30.952427Z","shell.execute_reply.started":"2023-03-01T21:18:17.663767Z","shell.execute_reply":"2023-03-01T21:18:30.951391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:30.955076Z","iopub.execute_input":"2023-03-01T21:18:30.955962Z","iopub.status.idle":"2023-03-01T21:18:30.981713Z","shell.execute_reply.started":"2023-03-01T21:18:30.955925Z","shell.execute_reply":"2023-03-01T21:18:30.980707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:30.983297Z","iopub.execute_input":"2023-03-01T21:18:30.983654Z","iopub.status.idle":"2023-03-01T21:18:31.461066Z","shell.execute_reply.started":"2023-03-01T21:18:30.983620Z","shell.execute_reply":"2023-03-01T21:18:31.460105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:31.462593Z","iopub.execute_input":"2023-03-01T21:18:31.463250Z","iopub.status.idle":"2023-03-01T21:18:31.468982Z","shell.execute_reply.started":"2023-03-01T21:18:31.463212Z","shell.execute_reply":"2023-03-01T21:18:31.468082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:31.470862Z","iopub.execute_input":"2023-03-01T21:18:31.471871Z","iopub.status.idle":"2023-03-01T21:18:31.480825Z","shell.execute_reply.started":"2023-03-01T21:18:31.471814Z","shell.execute_reply":"2023-03-01T21:18:31.479769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n_train_df = feature_engineer(train_df)\nprint( _train_df.shape )\n_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:31.482270Z","iopub.execute_input":"2023-03-01T21:18:31.482663Z","iopub.status.idle":"2023-03-01T21:18:46.294427Z","shell.execute_reply.started":"2023-03-01T21:18:31.482625Z","shell.execute_reply":"2023-03-01T21:18:46.293373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Manage Level groups \nFEATURES = [c for c in _train_df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = _train_df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:46.295979Z","iopub.execute_input":"2023-03-01T21:18:46.296406Z","iopub.status.idle":"2023-03-01T21:18:46.306406Z","shell.execute_reply.started":"2023-03-01T21:18:46.296371Z","shell.execute_reply":"2023-03-01T21:18:46.305162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf_xgb = GroupKFold(n_splits=5)\noof_xgb = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodel_xgb = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf_xgb.split(X=_train_df, groups=_train_df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = _train_df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = _train_df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf_xgb =  XGBClassifier(**xgb_params)\n        clf_xgb.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf_xgb.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        model_xgb[f'{grp}_{t}'] = clf_xgb\n        oof_xgb.loc[valid_users, t-1] = clf_xgb.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:18:46.309691Z","iopub.execute_input":"2023-03-01T21:18:46.309966Z","iopub.status.idle":"2023-03-01T21:19:26.227386Z","shell.execute_reply.started":"2023-03-01T21:18:46.309942Z","shell.execute_reply":"2023-03-01T21:19:26.226584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue_xgb = oof_xgb.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true_xgb[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:19:26.230678Z","iopub.execute_input":"2023-03-01T21:19:26.232699Z","iopub.status.idle":"2023-03-01T21:19:26.294343Z","shell.execute_reply.started":"2023-03-01T21:19:26.232666Z","shell.execute_reply":"2023-03-01T21:19:26.293362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscore_xgb = []; threshold_xgb = []\nbest_score_xgb = 0; best_threshold_xgb = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    pred_xgb = (oof_xgb.values.reshape((-1))>threshold).astype('int')\n    m_xgb = f1_score(true_xgb.values.reshape((-1)), pred_xgb, average='macro')   \n    score_xgb.append(m_xgb)\n    threshold_xgb.append(threshold)\n    if m_xgb > best_score_xgb:\n        best_score_xgb = m_xgb\n        best_threshold_xgb = threshold","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:19:26.295678Z","iopub.execute_input":"2023-03-01T21:19:26.297714Z","iopub.status.idle":"2023-03-01T21:19:28.892469Z","shell.execute_reply.started":"2023-03-01T21:19:26.297676Z","shell.execute_reply":"2023-03-01T21:19:28.890694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m_xgb = f1_score(true_xgb[k].values, (oof_xgb[k].values>best_threshold_xgb).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m_xgb)\n    \n# COMPUTE F1 SCORE OVERALL\nm_xgb = f1_score(true_xgb.values.reshape((-1)), (oof_xgb.values.reshape((-1))>best_threshold_xgb).astype('int'), average='macro')\nprint('==> Overall F1 =',m_xgb)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:19:28.896212Z","iopub.execute_input":"2023-03-01T21:19:28.896496Z","iopub.status.idle":"2023-03-01T21:19:29.028671Z","shell.execute_reply.started":"2023-03-01T21:19:28.896469Z","shell.execute_reply":"2023-03-01T21:19:29.027373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAR MEMORY\nimport gc\ndel oof_xgb, true_xgb\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:19:29.030662Z","iopub.execute_input":"2023-03-01T21:19:29.031680Z","iopub.status.idle":"2023-03-01T21:19:29.156591Z","shell.execute_reply.started":"2023-03-01T21:19:29.031641Z","shell.execute_reply":"2023-03-01T21:19:29.155429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:19:29.158103Z","iopub.execute_input":"2023-03-01T21:19:29.160034Z","iopub.status.idle":"2023-03-01T21:19:29.183714Z","shell.execute_reply.started":"2023-03-01T21:19:29.159999Z","shell.execute_reply":"2023-03-01T21:19:29.182837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n  \nfor (sample_submission, test) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf_xgb = model_xgb[f'{grp}_{t}']\n        p = clf_xgb.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold_xgb)\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:19:29.185101Z","iopub.execute_input":"2023-03-01T21:19:29.185451Z","iopub.status.idle":"2023-03-01T21:19:29.657834Z","shell.execute_reply.started":"2023-03-01T21:19:29.185416Z","shell.execute_reply":"2023-03-01T21:19:29.657095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:23:15.068976Z","iopub.execute_input":"2023-03-01T21:23:15.069362Z","iopub.status.idle":"2023-03-01T21:23:15.084697Z","shell.execute_reply.started":"2023-03-01T21:23:15.069330Z","shell.execute_reply":"2023-03-01T21:23:15.083591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"execution":{"iopub.status.busy":"2023-03-01T21:23:59.515890Z","iopub.execute_input":"2023-03-01T21:23:59.516262Z","iopub.status.idle":"2023-03-01T21:23:59.522456Z","shell.execute_reply.started":"2023-03-01T21:23:59.516232Z","shell.execute_reply":"2023-03-01T21:23:59.521267Z"},"trusted":true},"execution_count":null,"outputs":[]}]}