{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\n\npd.set_option('display.max_columns', None)","metadata":{"executionInfo":{"elapsed":6000,"status":"ok","timestamp":1686558944539,"user":{"displayName":"Belajar Bersama","userId":"00009692112787889038"},"user_tz":-420},"id":"5Q2fcroqSbbz","execution":{"iopub.status.busy":"2023-06-20T21:09:02.809428Z","iopub.execute_input":"2023-06-20T21:09:02.809968Z","iopub.status.idle":"2023-06-20T21:09:04.291040Z","shell.execute_reply.started":"2023-06-20T21:09:02.809945Z","shell.execute_reply":"2023-06-20T21:09:04.289946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':np.int32,\n    'hq':np.int32,\n    'music':np.int32,\n    'level_group':'category'}\n\ntrain = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:09:04.295293Z","iopub.execute_input":"2023-06-20T21:09:04.295631Z","iopub.status.idle":"2023-06-20T21:10:26.149215Z","shell.execute_reply.started":"2023-06-20T21:09:04.295603Z","shell.execute_reply":"2023-06-20T21:10:26.148515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\ntrain_labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"id":"vZVk1W9NTG2G","execution":{"iopub.status.busy":"2023-06-20T21:10:26.150634Z","iopub.execute_input":"2023-06-20T21:10:26.151171Z","iopub.status.idle":"2023-06-20T21:10:26.478888Z","shell.execute_reply.started":"2023-06-20T21:10:26.151148Z","shell.execute_reply":"2023-06-20T21:10:26.477796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Overview","metadata":{"id":"nn-TtbErX9jY"}},{"cell_type":"code","source":"train.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:26.480924Z","iopub.execute_input":"2023-06-20T21:10:26.482022Z","iopub.status.idle":"2023-06-20T21:10:26.516336Z","shell.execute_reply.started":"2023-06-20T21:10:26.481978Z","shell.execute_reply":"2023-06-20T21:10:26.515472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:26.517285Z","iopub.execute_input":"2023-06-20T21:10:26.517525Z","iopub.status.idle":"2023-06-20T21:10:26.524296Z","shell.execute_reply.started":"2023-06-20T21:10:26.517506Z","shell.execute_reply":"2023-06-20T21:10:26.523194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:26.525663Z","iopub.execute_input":"2023-06-20T21:10:26.526115Z","iopub.status.idle":"2023-06-20T21:10:26.558467Z","shell.execute_reply.started":"2023-06-20T21:10:26.526087Z","shell.execute_reply":"2023-06-20T21:10:26.557773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"train_labels['quest'] = train_labels['session_id'].apply(lambda x : x.split('_')[-1].split('q')[1])\ntrain_labels['Session'] = train_labels['session_id'].apply(lambda x : x.split('_')[0]).astype('Int64')\ndfg = train_labels.groupby('quest').sum().reset_index()\ndfg['quest'] = dfg['quest'].apply(lambda x: x.split('q')[-1])\ndfg['quest'] = dfg['quest'].astype(int)\ndfg = dfg.sort_values('quest')\ntrain_labels['quest'] = train_labels['quest'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:26.559566Z","iopub.execute_input":"2023-06-20T21:10:26.560013Z","iopub.status.idle":"2023-06-20T21:10:27.044492Z","shell.execute_reply.started":"2023-06-20T21:10:26.559987Z","shell.execute_reply":"2023-06-20T21:10:27.043770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:27.045550Z","iopub.execute_input":"2023-06-20T21:10:27.045892Z","iopub.status.idle":"2023-06-20T21:10:27.056020Z","shell.execute_reply.started":"2023-06-20T21:10:27.045872Z","shell.execute_reply":"2023-06-20T21:10:27.055114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nax = sns.barplot(x = 'quest', y = 'correct', data = dfg, palette='flare')\nplt.bar_label(ax.containers[0])\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:27.057283Z","iopub.execute_input":"2023-06-20T21:10:27.057522Z","iopub.status.idle":"2023-06-20T21:10:27.400618Z","shell.execute_reply.started":"2023-06-20T21:10:27.057501Z","shell.execute_reply":"2023-06-20T21:10:27.399526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"df = train.copy().groupby(['session_id','level_group']).aggregate({\n    'elapsed_time' : 'sum'\n    ,'event_name' :\t'nunique'\n    ,'name'\t: 'nunique'\n    ,'level' : 'mean'\n    ,'room_coor_x' : 'mean'\n    ,'room_coor_y' : 'mean'\n    ,'screen_coor_x' : 'mean'\n    ,'screen_coor_y' : 'mean'\n    ,'hover_duration' : 'mean'\n    ,'text' : 'nunique'\n    ,'fqid' : 'nunique'\n    ,'room_fqid' : 'nunique'\n    ,'text_fqid' : 'nunique'\n    ,'fullscreen' : 'mean'\n    ,'hq' : 'mean'\n    ,'music' : 'mean'\n}).reset_index()\ndf = df.set_index('session_id')\ndf['hover_duration'] = df['hover_duration'].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:27.403751Z","iopub.execute_input":"2023-06-20T21:10:27.404269Z","iopub.status.idle":"2023-06-20T21:10:42.947379Z","shell.execute_reply.started":"2023-06-20T21:10:27.404225Z","shell.execute_reply":"2023-06-20T21:10:42.946657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_USERS = df.index.unique()\nFEATURES = [c for c in df.columns if not c in ['level_group']]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:42.948559Z","iopub.execute_input":"2023-06-20T21:10:42.948995Z","iopub.status.idle":"2023-06-20T21:10:42.956055Z","shell.execute_reply.started":"2023-06-20T21:10:42.948974Z","shell.execute_reply":"2023-06-20T21:10:42.955457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score\nimport joblib\n\ngkf = GroupKFold(n_splits = 5)\noof = pd.DataFrame(data = np.zeros((len(ALL_USERS), 18)), index = ALL_USERS)\nscores = pd.DataFrame(\n    index = [f'FOLD_{i}' for i in range(5)],\n    columns = list(range(1, 19))\n)\nmodels = {}\n\nfor i, (t, v) in enumerate(gkf.split(X = df, groups = df.index)) :\n    print(f\"FOLD {i}\")\n    print('')\n    \n    for l in range(1, 19) :\n        if l <= 4 : grp = '0-4'\n        if l <= 12 : grp = '5-12'\n        if l <= 22 : grp = '13-22'\n        \n        X_train = df.iloc[t]\n        X_train = X_train.loc[X_train.level_group == grp]\n        train_users = X_train.index.values\n        y_train = train_labels.loc[train_labels.quest==l].set_index('Session').loc[train_users]\n        \n        X_test = df.iloc[v]\n        X_test = X_test.loc[X_test.level_group == grp]\n        val_users = X_test.index.values\n        y_test = train_labels.loc[train_labels.quest==l].set_index('Session').loc[val_users]        \n        \n        model = RandomForestClassifier(random_state=42, criterion='entropy', max_depth=8,max_features='sqrt',min_samples_leaf=10,min_samples_split=14,n_estimators=147)\n        model.fit( X_train[FEATURES].astype(np.float32), y_train['correct'])\n        \n        yhat = model.predict_proba(X_test[FEATURES].astype(np.float32))[:, 1]\n        score = f1_score(y_test['correct'], np.round(yhat).astype(int))\n        print(f'Quest {l} : {score}')\n        models[f'fold{i}-level{l}'] = model\n        joblib.dump(model, f'model-fold{i}-level{l}.pkl')\n        oof.loc[val_users, l-1] = yhat\n        scores.loc[f'FOLD_{i}', l] = score\n        del X_train, train_users, y_train, X_test, val_users, y_test, model, yhat, score\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:10:42.957262Z","iopub.execute_input":"2023-06-20T21:10:42.957695Z","iopub.status.idle":"2023-06-20T21:19:17.141846Z","shell.execute_reply.started":"2023-06-20T21:10:42.957674Z","shell.execute_reply":"2023-06-20T21:19:17.140331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores.mean()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:17.143664Z","iopub.execute_input":"2023-06-20T21:19:17.143978Z","iopub.status.idle":"2023-06-20T21:19:17.155681Z","shell.execute_reply.started":"2023-06-20T21:19:17.143952Z","shell.execute_reply":"2023-06-20T21:19:17.154127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores.mean().mean()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:17.157421Z","iopub.execute_input":"2023-06-20T21:19:17.157784Z","iopub.status.idle":"2023-06-20T21:19:17.168227Z","shell.execute_reply.started":"2023-06-20T21:19:17.157757Z","shell.execute_reply":"2023-06-20T21:19:17.167181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_labels.loc[train_labels.quest == k+1].set_index('Session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:17.169181Z","iopub.execute_input":"2023-06-20T21:19:17.170080Z","iopub.status.idle":"2023-06-20T21:19:17.460210Z","shell.execute_reply.started":"2023-06-20T21:19:17.170039Z","shell.execute_reply":"2023-06-20T21:19:17.458620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:17.461318Z","iopub.execute_input":"2023-06-20T21:19:17.461669Z","iopub.status.idle":"2023-06-20T21:19:24.452161Z","shell.execute_reply.started":"2023-06-20T21:19:17.461643Z","shell.execute_reply":"2023-06-20T21:19:24.451492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:24.455638Z","iopub.execute_input":"2023-06-20T21:19:24.455950Z","iopub.status.idle":"2023-06-20T21:19:24.701421Z","shell.execute_reply.started":"2023-06-20T21:19:24.455928Z","shell.execute_reply":"2023-06-20T21:19:24.699678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:24.702911Z","iopub.execute_input":"2023-06-20T21:19:24.704707Z","iopub.status.idle":"2023-06-20T21:19:24.937297Z","shell.execute_reply.started":"2023-06-20T21:19:24.704670Z","shell.execute_reply":"2023-06-20T21:19:24.936017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:24.938739Z","iopub.execute_input":"2023-06-20T21:19:24.939093Z","iopub.status.idle":"2023-06-20T21:19:24.974103Z","shell.execute_reply.started":"2023-06-20T21:19:24.939065Z","shell.execute_reply":"2023-06-20T21:19:24.973360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAR MEMORY\nimport gc\ndel train_labels, oof, true\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:24.975240Z","iopub.execute_input":"2023-06-20T21:19:24.976101Z","iopub.status.idle":"2023-06-20T21:19:25.127348Z","shell.execute_reply.started":"2023-06-20T21:19:24.976078Z","shell.execute_reply":"2023-06-20T21:19:25.126573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\nFOLD_TO_USE = 2\n\nfor (test, sample_submission) in iter_test:\n    \n    test = test.groupby(['session_id','level_group']).aggregate({\n    'elapsed_time' : 'sum'\n    ,'event_name' :\t'nunique'\n    ,'name'\t: 'nunique'\n    ,'level' : 'mean'\n    ,'room_coor_x' : 'mean'\n    ,'room_coor_y' : 'mean'\n    ,'screen_coor_x' : 'mean'\n    ,'screen_coor_y' : 'mean'\n    ,'hover_duration' : 'mean'\n    ,'text' : 'nunique'\n    ,'fqid' : 'nunique'\n    ,'room_fqid' : 'nunique'\n    ,'text_fqid' : 'nunique'\n    ,'fullscreen' : 'mean'\n    ,'hq' : 'mean'\n    ,'music' : 'mean'\n    }).reset_index()\n    test = test.set_index('session_id')\n    test['hover_duration'] = test['hover_duration'].fillna(0)\n        \n    # INFER TEST DATA\n    grp = '0-4'\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'fold{FOLD_TO_USE}-level{t}']\n        p = clf.predict_proba(test[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:19:25.128735Z","iopub.execute_input":"2023-06-20T21:19:25.129635Z","iopub.status.idle":"2023-06-20T21:19:25.687065Z","shell.execute_reply.started":"2023-06-20T21:19:25.129607Z","shell.execute_reply":"2023-06-20T21:19:25.685925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv').head(10)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:51:41.521119Z","iopub.execute_input":"2023-06-20T21:51:41.521525Z","iopub.status.idle":"2023-06-20T21:51:41.535960Z","shell.execute_reply.started":"2023-06-20T21:51:41.521502Z","shell.execute_reply":"2023-06-20T21:51:41.534746Z"},"trusted":true},"execution_count":null,"outputs":[]}]}