{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:16:29.135838Z","iopub.execute_input":"2023-04-30T00:16:29.136351Z","iopub.status.idle":"2023-04-30T00:16:29.142269Z","shell.execute_reply.started":"2023-04-30T00:16:29.136309Z","shell.execute_reply":"2023-04-30T00:16:29.141083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')\n\n# COMPUTE READS AND SKIPS\nPIECES = 10\nCHUNK = int( np.ceil(len(tmp)/PIECES) )\n\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k*CHUNK\n    b = (k+1)*CHUNK\n    if b>len(tmp): b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)\n    \nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:16:29.544759Z","iopub.execute_input":"2023-04-30T00:16:29.545231Z","iopub.status.idle":"2023-04-30T00:16:57.861991Z","shell.execute_reply.started":"2023-04-30T00:16:29.545187Z","shell.execute_reply":"2023-04-30T00:16:57.860643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[0])\nprint('Train size of first piece:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:12.746431Z","iopub.execute_input":"2023-04-30T00:20:12.747485Z","iopub.status.idle":"2023-04-30T00:20:19.911925Z","shell.execute_reply.started":"2023-04-30T00:20:12.747431Z","shell.execute_reply":"2023-04-30T00:20:19.910944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:19.914016Z","iopub.execute_input":"2023-04-30T00:20:19.914368Z","iopub.status.idle":"2023-04-30T00:20:19.922143Z","shell.execute_reply.started":"2023-04-30T00:20:19.914333Z","shell.execute_reply":"2023-04-30T00:20:19.921106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:19.923782Z","iopub.execute_input":"2023-04-30T00:20:19.924182Z","iopub.status.idle":"2023-04-30T00:20:19.949221Z","shell.execute_reply.started":"2023-04-30T00:20:19.924132Z","shell.execute_reply":"2023-04-30T00:20:19.948039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ny['session'] = y.session_id.apply(lambda x: int(x.split('_')[0]) )\ny['q'] = y.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:25.749075Z","iopub.execute_input":"2023-04-30T00:20:25.749490Z","iopub.status.idle":"2023-04-30T00:20:26.764160Z","shell.execute_reply.started":"2023-04-30T00:20:25.749456Z","shell.execute_reply":"2023-04-30T00:20:26.762999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:26.765865Z","iopub.execute_input":"2023-04-30T00:20:26.766562Z","iopub.status.idle":"2023-04-30T00:20:26.772760Z","shell.execute_reply.started":"2023-04-30T00:20:26.766519Z","shell.execute_reply":"2023-04-30T00:20:26.771829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:26.822805Z","iopub.execute_input":"2023-04-30T00:20:26.823507Z","iopub.status.idle":"2023-04-30T00:20:26.834440Z","shell.execute_reply.started":"2023-04-30T00:20:26.823463Z","shell.execute_reply":"2023-04-30T00:20:26.833237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"temp_evntNames = pd.get_dummies(train['event_name'])\n\ntrain = pd.concat([train, temp_evntNames], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:30.159489Z","iopub.execute_input":"2023-04-30T00:20:30.159938Z","iopub.status.idle":"2023-04-30T00:20:30.775151Z","shell.execute_reply.started":"2023-04-30T00:20:30.159881Z","shell.execute_reply":"2023-04-30T00:20:30.773818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORIES = ['event_name', 'name','fqid', 'room_fqid', 'text']\nMEAN_VAR = ['elapsed_time','level','page','room_coor_x', 'room_coor_y','screen_coor_x', 'screen_coor_y', 'hover_duration']\nEVENT_VAR = ['navigate_click','person_click','cutscene_click','object_click','map_hover','notification_click',\n            'map_click','observation_click','checkpoint']#,'elapsed_time']","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:31.453812Z","iopub.execute_input":"2023-04-30T00:20:31.454257Z","iopub.status.idle":"2023-04-30T00:20:31.460336Z","shell.execute_reply.started":"2023-04-30T00:20:31.454216Z","shell.execute_reply":"2023-04-30T00:20:31.459241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATEGORIES:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in MEAN_VAR:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in MEAN_VAR:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENT_VAR: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENT_VAR + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENT_VAR,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:32.389741Z","iopub.execute_input":"2023-04-30T00:20:32.390621Z","iopub.status.idle":"2023-04-30T00:20:32.401372Z","shell.execute_reply.started":"2023-04-30T00:20:32.390571Z","shell.execute_reply":"2023-04-30T00:20:32.399971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineer(train)\n    all_pieces.append(df)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train; \ngc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:20:36.488332Z","iopub.execute_input":"2023-04-30T00:20:36.488747Z","iopub.status.idle":"2023-04-30T00:26:51.100504Z","shell.execute_reply.started":"2023-04-30T00:20:36.488709Z","shell.execute_reply":"2023-04-30T00:26:51.099444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nALL_USERS = df.index.unique()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:27:48.580743Z","iopub.execute_input":"2023-04-30T00:27:48.581244Z","iopub.status.idle":"2023-04-30T00:27:48.593275Z","shell.execute_reply.started":"2023-04-30T00:27:48.581197Z","shell.execute_reply":"2023-04-30T00:27:48.591927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:27:49.564394Z","iopub.execute_input":"2023-04-30T00:27:49.565568Z","iopub.status.idle":"2023-04-30T00:27:49.573069Z","shell.execute_reply.started":"2023-04-30T00:27:49.565514Z","shell.execute_reply":"2023-04-30T00:27:49.571916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_USERS","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:27:50.779243Z","iopub.execute_input":"2023-04-30T00:27:50.780128Z","iopub.status.idle":"2023-04-30T00:27:50.787173Z","shell.execute_reply.started":"2023-04-30T00:27:50.780081Z","shell.execute_reply":"2023-04-30T00:27:50.786077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MODELING","metadata":{}},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'evail_metric' : 'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 5,\n    'num_iterations': 1000}\n    \n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = y.loc[y.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = y.loc[y.q==t].set_index('session').loc[valid_users]\n        \n        #TRAIN MODEL\n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:27:54.081251Z","iopub.execute_input":"2023-04-30T00:27:54.081752Z","iopub.status.idle":"2023-04-30T00:34:50.829246Z","shell.execute_reply.started":"2023-04-30T00:27:54.081706Z","shell.execute_reply":"2023-04-30T00:34:50.827727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = y.loc[y.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:35:57.701440Z","iopub.execute_input":"2023-04-30T00:35:57.701901Z","iopub.status.idle":"2023-04-30T00:35:57.827576Z","shell.execute_reply.started":"2023-04-30T00:35:57.701848Z","shell.execute_reply":"2023-04-30T00:35:57.826328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:36:00.511821Z","iopub.execute_input":"2023-04-30T00:36:00.513239Z","iopub.status.idle":"2023-04-30T00:36:07.015199Z","shell.execute_reply.started":"2023-04-30T00:36:00.513173Z","shell.execute_reply":"2023-04-30T00:36:07.013862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:36:07.020293Z","iopub.execute_input":"2023-04-30T00:36:07.021223Z","iopub.status.idle":"2023-04-30T00:36:07.363214Z","shell.execute_reply.started":"2023-04-30T00:36:07.021167Z","shell.execute_reply":"2023-04-30T00:36:07.362211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### INFER TEST DATA","metadata":{}},{"cell_type":"code","source":"#IMPORTING API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# # CLEAR MEM\n# import gc\n# del test, df, oof, true\n# _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:36:15.302573Z","iopub.execute_input":"2023-04-30T00:36:15.303025Z","iopub.status.idle":"2023-04-30T00:36:15.446287Z","shell.execute_reply.started":"2023-04-30T00:36:15.302983Z","shell.execute_reply":"2023-04-30T00:36:15.444452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:36:23.767082Z","iopub.execute_input":"2023-04-30T00:36:23.767542Z","iopub.status.idle":"2023-04-30T00:36:24.724094Z","shell.execute_reply.started":"2023-04-30T00:36:23.767502Z","shell.execute_reply":"2023-04-30T00:36:24.722796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:37:24.460251Z","iopub.execute_input":"2023-04-30T00:37:24.460879Z","iopub.status.idle":"2023-04-30T00:37:24.477654Z","shell.execute_reply.started":"2023-04-30T00:37:24.460836Z","shell.execute_reply":"2023-04-30T00:37:24.476460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:37:39.025162Z","iopub.execute_input":"2023-04-30T00:37:39.025605Z","iopub.status.idle":"2023-04-30T00:37:39.033469Z","shell.execute_reply.started":"2023-04-30T00:37:39.025562Z","shell.execute_reply":"2023-04-30T00:37:39.032009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:38:50.478067Z","iopub.execute_input":"2023-04-30T00:38:50.478563Z","iopub.status.idle":"2023-04-30T00:38:50.487413Z","shell.execute_reply.started":"2023-04-30T00:38:50.478520Z","shell.execute_reply":"2023-04-30T00:38:50.486108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit","metadata":{"execution":{"iopub.status.busy":"2023-04-30T00:38:53.809948Z","iopub.execute_input":"2023-04-30T00:38:53.810395Z","iopub.status.idle":"2023-04-30T00:38:53.828913Z","shell.execute_reply.started":"2023-04-30T00:38:53.810355Z","shell.execute_reply":"2023-04-30T00:38:53.826044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}