{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import KFold, GroupKFold","metadata":{"papermill":{"duration":1.504459,"end_time":"2023-02-08T02:45:45.143437","exception":false,"start_time":"2023-02-08T02:45:43.638978","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:46:40.631997Z","iopub.execute_input":"2023-02-11T13:46:40.632330Z","iopub.status.idle":"2023-02-11T13:46:42.196334Z","shell.execute_reply.started":"2023-02-11T13:46:40.632255Z","shell.execute_reply":"2023-02-11T13:46:42.195195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\nprint( train.shape )\ntrain.head()","metadata":{"papermill":{"duration":67.756839,"end_time":"2023-02-08T02:46:52.917276","exception":false,"start_time":"2023-02-08T02:45:45.160437","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:46:42.198543Z","iopub.execute_input":"2023-02-11T13:46:42.199173Z","iopub.status.idle":"2023-02-11T13:47:37.597401Z","shell.execute_reply.started":"2023-02-11T13:46:42.199131Z","shell.execute_reply":"2023-02-11T13:47:37.596225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"papermill":{"duration":0.69016,"end_time":"2023-02-08T02:46:53.614270","exception":false,"start_time":"2023-02-08T02:46:52.924110","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:47:37.599232Z","iopub.execute_input":"2023-02-11T13:47:37.599998Z","iopub.status.idle":"2023-02-11T13:47:38.139256Z","shell.execute_reply.started":"2023-02-11T13:47:37.599957Z","shell.execute_reply":"2023-02-11T13:47:38.138128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"papermill":{"duration":0.022719,"end_time":"2023-02-08T02:46:53.663191","exception":false,"start_time":"2023-02-08T02:46:53.640472","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:47:38.143282Z","iopub.execute_input":"2023-02-11T13:47:38.143571Z","iopub.status.idle":"2023-02-11T13:47:38.150673Z","shell.execute_reply.started":"2023-02-11T13:47:38.143545Z","shell.execute_reply":"2023-02-11T13:47:38.149515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"papermill":{"duration":0.023977,"end_time":"2023-02-08T02:46:53.696198","exception":false,"start_time":"2023-02-08T02:46:53.672221","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:47:38.153202Z","iopub.execute_input":"2023-02-11T13:47:38.153487Z","iopub.status.idle":"2023-02-11T13:47:38.163146Z","shell.execute_reply.started":"2023-02-11T13:47:38.153461Z","shell.execute_reply":"2023-02-11T13:47:38.162024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":41.838,"end_time":"2023-02-08T02:47:35.543843","exception":false,"start_time":"2023-02-08T02:46:53.705843","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:47:38.164526Z","iopub.execute_input":"2023-02-11T13:47:38.164980Z","iopub.status.idle":"2023-02-11T13:48:17.474860Z","shell.execute_reply.started":"2023-02-11T13:47:38.164939Z","shell.execute_reply":"2023-02-11T13:48:17.473683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"papermill":{"duration":0.022765,"end_time":"2023-02-08T02:47:35.588161","exception":false,"start_time":"2023-02-08T02:47:35.565396","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:48:17.476525Z","iopub.execute_input":"2023-02-11T13:48:17.477511Z","iopub.status.idle":"2023-02-11T13:48:17.487807Z","shell.execute_reply.started":"2023-02-11T13:48:17.477470Z","shell.execute_reply":"2023-02-11T13:48:17.486684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        clf = RandomForestClassifier() \n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print('\\n')","metadata":{"papermill":{"duration":342.283608,"end_time":"2023-02-08T02:53:17.879196","exception":false,"start_time":"2023-02-08T02:47:35.595588","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:48:17.489502Z","iopub.execute_input":"2023-02-11T13:48:17.490036Z","iopub.status.idle":"2023-02-11T13:53:25.147122Z","shell.execute_reply.started":"2023-02-11T13:48:17.489996Z","shell.execute_reply":"2023-02-11T13:53:25.145124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"papermill":{"duration":0.108412,"end_time":"2023-02-08T02:53:18.025668","exception":false,"start_time":"2023-02-08T02:53:17.917256","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:25.148758Z","iopub.execute_input":"2023-02-11T13:53:25.149141Z","iopub.status.idle":"2023-02-11T13:53:25.215250Z","shell.execute_reply.started":"2023-02-11T13:53:25.149103Z","shell.execute_reply":"2023-02-11T13:53:25.213914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"papermill":{"duration":4.134012,"end_time":"2023-02-08T02:53:22.172310","exception":false,"start_time":"2023-02-08T02:53:18.038298","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:25.219375Z","iopub.execute_input":"2023-02-11T13:53:25.219776Z","iopub.status.idle":"2023-02-11T13:53:27.885404Z","shell.execute_reply.started":"2023-02-11T13:53:25.219740Z","shell.execute_reply":"2023-02-11T13:53:27.884239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"papermill":{"duration":0.288165,"end_time":"2023-02-08T02:53:22.474770","exception":false,"start_time":"2023-02-08T02:53:22.186605","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:27.887045Z","iopub.execute_input":"2023-02-11T13:53:27.887835Z","iopub.status.idle":"2023-02-11T13:53:28.184503Z","shell.execute_reply.started":"2023-02-11T13:53:27.887788Z","shell.execute_reply":"2023-02-11T13:53:28.183431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"papermill":{"duration":0.229092,"end_time":"2023-02-08T02:53:22.718707","exception":false,"start_time":"2023-02-08T02:53:22.489615","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:28.186073Z","iopub.execute_input":"2023-02-11T13:53:28.186729Z","iopub.status.idle":"2023-02-11T13:53:28.342677Z","shell.execute_reply.started":"2023-02-11T13:53:28.186685Z","shell.execute_reply":"2023-02-11T13:53:28.341165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"papermill":{"duration":0.06105,"end_time":"2023-02-08T02:53:22.824076","exception":false,"start_time":"2023-02-08T02:53:22.763026","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:28.344587Z","iopub.execute_input":"2023-02-11T13:53:28.345065Z","iopub.status.idle":"2023-02-11T13:53:28.374419Z","shell.execute_reply.started":"2023-02-11T13:53:28.345022Z","shell.execute_reply":"2023-02-11T13:53:28.373424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (sample_submission, test) in iter_test:\n    \n    df = feature_engineer(test)\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold)\n    \n    env.predict(sample_submission)","metadata":{"papermill":{"duration":1.199267,"end_time":"2023-02-08T02:53:24.038072","exception":false,"start_time":"2023-02-08T02:53:22.838805","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:28.376765Z","iopub.execute_input":"2023-02-11T13:53:28.377934Z","iopub.status.idle":"2023-02-11T13:53:29.366526Z","shell.execute_reply.started":"2023-02-11T13:53:28.377872Z","shell.execute_reply":"2023-02-11T13:53:29.365559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":0.03378,"end_time":"2023-02-08T02:53:24.115929","exception":false,"start_time":"2023-02-08T02:53:24.082149","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:29.368033Z","iopub.execute_input":"2023-02-11T13:53:29.368489Z","iopub.status.idle":"2023-02-11T13:53:29.383089Z","shell.execute_reply.started":"2023-02-11T13:53:29.368445Z","shell.execute_reply":"2023-02-11T13:53:29.382010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"papermill":{"duration":0.025602,"end_time":"2023-02-08T02:53:24.156618","exception":false,"start_time":"2023-02-08T02:53:24.131016","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-11T13:53:29.384602Z","iopub.execute_input":"2023-02-11T13:53:29.384957Z","iopub.status.idle":"2023-02-11T13:53:29.391223Z","shell.execute_reply.started":"2023-02-11T13:53:29.384924Z","shell.execute_reply":"2023-02-11T13:53:29.390058Z"},"trusted":true},"execution_count":null,"outputs":[]}]}