{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import f1_score","metadata":{"papermill":{"duration":1.027875,"end_time":"2023-02-07T00:59:59.180261","exception":false,"start_time":"2023-02-07T00:59:58.152386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:00:40.968574Z","iopub.execute_input":"2023-05-05T14:00:40.968988Z","iopub.status.idle":"2023-05-05T14:00:40.974954Z","shell.execute_reply.started":"2023-05-05T14:00:40.968957Z","shell.execute_reply":"2023-05-05T14:00:40.973684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels\nOn March 20 2023, Kaggle doubled the size of train data (discussion [here][1]). The train data is now 4.7GB! To avoid memory error, we will read the train data in as 10 pieces and feature engineer each piece before reading the next piece. This works because feature engineering shrinks the size of each piece.\n\n[1]: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/396202","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"user = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=['session_id'])\nuser = user.groupby('session_id').session_id.agg('count')\n\npcs = 20\nc = int(np.ceil(len(user)/pcs))\n\nbaca = []\nlewat = [0]\n\nfor i in range(pcs):\n    a = i*c\n    b = (i+1)*c\n    if b > len(user): b = len(user)\n    rows = user.iloc[a:b].sum()\n    baca.append(rows)\n    lewat.append(lewat[-1] + rows)\n    \n\nprint(f' we will read the train.csv file in {pcs} pieces of sizes:')\nprint(baca)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-05T14:00:41.044449Z","iopub.execute_input":"2023-05-05T14:00:41.044888Z","iopub.status.idle":"2023-05-05T14:01:10.079466Z","shell.execute_reply.started":"2023-05-05T14:00:41.044853Z","shell.execute_reply":"2023-05-05T14:01:10.078191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',nrows=baca[0])\nprint(tmp.shape)\ntmp.head(3)","metadata":{"papermill":{"duration":59.284316,"end_time":"2023-02-07T01:00:58.478743","exception":false,"start_time":"2023-02-07T00:59:59.194427","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:01:10.081588Z","iopub.execute_input":"2023-05-05T14:01:10.082101Z","iopub.status.idle":"2023-05-05T14:01:14.017746Z","shell.execute_reply.started":"2023-05-05T14:01:10.082064Z","shell.execute_reply":"2023-05-05T14:01:14.016661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_label = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ndata_label['session'] = data_label.session_id.apply(lambda x: int(x.split('_')[0]))\ndata_label['q'] = data_label.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\n\ndata_label['correct'] = data_label['correct'].astype(bool)\nprint(data_label.shape)\ndata_label.head()","metadata":{"papermill":{"duration":0.598155,"end_time":"2023-02-07T01:00:59.082015","exception":false,"start_time":"2023-02-07T01:00:58.48386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:01:14.019136Z","iopub.execute_input":"2023-05-05T14:01:14.019736Z","iopub.status.idle":"2023-05-05T14:01:15.283443Z","shell.execute_reply.started":"2023-05-05T14:01:14.019698Z","shell.execute_reply":"2023-05-05T14:01:15.282547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time', 'level', 'page', 'room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\nEVENTS = ['navigate_click', 'person_click', 'cutscene_click', 'object_click',\n          'map_hover', 'notification_click', 'map_click', 'observation_click',\n          'checkpoint']","metadata":{"papermill":{"duration":0.014685,"end_time":"2023-02-07T01:00:59.112856","exception":false,"start_time":"2023-02-07T01:00:59.098171","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:01:15.285722Z","iopub.execute_input":"2023-05-05T14:01:15.286351Z","iopub.status.idle":"2023-05-05T14:01:15.291439Z","shell.execute_reply.started":"2023-05-05T14:01:15.286312Z","shell.execute_reply":"2023-05-05T14:01:15.290566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fitur_enginer(train):\n    # Create an empty list to store DataFrames\n    dfs = []\n    \n\n    for c in CATS:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('nunique')\n        tmporary.name = tmporary.name + '_nunique'\n        dfs.append(tmporary)\n        \n\n    for c in NUMS:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('mean')\n        tmporary.name = tmporary.name + '_mean'\n        dfs.append(tmporary)\n    for c in NUMS:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('std')\n        tmporary.name = tmporary.name + '_std'\n        dfs.append(tmporary)\n        \n\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n        \n\n    for c in EVENTS + ['elapsed_time']:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('sum')\n        tmp.name = tmporary.name + '_sum'\n        dfs.append(tmporary)\n        \n        \n    train = train.drop(EVENTS, axis=1)\n    df = pd.concat(dfs, axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"papermill":{"duration":0.017716,"end_time":"2023-02-07T01:00:59.136021","exception":false,"start_time":"2023-02-07T01:00:59.118305","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:01:15.292781Z","iopub.execute_input":"2023-05-05T14:01:15.293316Z","iopub.status.idle":"2023-05-05T14:01:15.306782Z","shell.execute_reply.started":"2023-05-05T14:01:15.293283Z","shell.execute_reply":"2023-05-05T14:01:15.305776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Process train data in pieces to avoid memory error\nall_pieces = []\nprint(f'Processing train as {pcs} pieces to avoid memory error... ')\nfor i in range(pcs):\n    print(i, ', ', end='')\n    SKIPS = 0\n    if i > 0:\n        SKIPS = range(1, lewat[i] + 1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=baca[i], skiprows=SKIPS)\n    df = fitur_enginer(train)\n    all_pieces.append(df)\n\n# Concatenate all pieces\nprint('\\n')\ndel train\ngc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape)\ndf.head()","metadata":{"papermill":{"duration":34.516494,"end_time":"2023-02-07T01:01:33.658043","exception":false,"start_time":"2023-02-07T01:00:59.141549","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:01:15.308160Z","iopub.execute_input":"2023-05-05T14:01:15.308721Z","iopub.status.idle":"2023-05-05T14:10:29.903717Z","shell.execute_reply.started":"2023-05-05T14:01:15.308688Z","shell.execute_reply":"2023-05-05T14:10:29.902224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\n\n# Dapat kan ID unik\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"papermill":{"duration":0.014699,"end_time":"2023-02-07T01:01:33.689953","exception":false,"start_time":"2023-02-07T01:01:33.675254","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:10:29.906274Z","iopub.execute_input":"2023-05-05T14:10:29.906758Z","iopub.status.idle":"2023-05-05T14:10:29.920341Z","shell.execute_reply.started":"2023-05-05T14:10:29.906718Z","shell.execute_reply":"2023-05-05T14:10:29.919126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Definisikan fitur dan print nilai fitur dan user\nFEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')\n\ngkf = GroupKFold(n_splits=5)\n\n# Inisiasi oof prediksi dataframe dan dictionari model\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# Compute Cv Score With 5 Group K Fold\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False }\n    \n    # Iterasi\n    for t in range(1,19):\n        \n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # Train Data\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = data_label.loc[data_label.q==t].set_index('session').loc[train_users]\n        \n        # Valid Data\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = data_label.loc[data_label.q==t].set_index('session').loc[valid_users]\n        \n        # Train Model\n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[(valid_x[FEATURES].astype('float32'), valid_y['correct'])],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # Save Model, Predict Valid OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"papermill":{"duration":69.877213,"end_time":"2023-02-07T01:02:43.57299","exception":false,"start_time":"2023-02-07T01:01:33.695777","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:10:29.922119Z","iopub.execute_input":"2023-05-05T14:10:29.922516Z","iopub.status.idle":"2023-05-05T14:13:17.008688Z","shell.execute_reply.started":"2023-05-05T14:10:29.922472Z","shell.execute_reply":"2023-05-05T14:13:17.007644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true = oof.copy()\nfor i in range(18):\n    tmporary = data_label.loc[data_label.q == i+1].set_index('session').loc[ALL_USERS]\n    true[i] = tmporary.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-05-05T14:13:17.013031Z","iopub.execute_input":"2023-05-05T14:13:17.015288Z","iopub.status.idle":"2023-05-05T14:13:17.155537Z","shell.execute_reply.started":"2023-05-05T14:13:17.015244Z","shell.execute_reply":"2023-05-05T14:13:17.154544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []; thresholds = []\nbest_score = 0\nbest_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-05T14:13:17.160431Z","iopub.execute_input":"2023-05-05T14:13:17.160862Z","iopub.status.idle":"2023-05-05T14:13:25.068443Z","shell.execute_reply.started":"2023-05-05T14:13:17.160829Z","shell.execute_reply":"2023-05-05T14:13:25.066927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='green')\nplt.scatter([best_threshold], [best_score], color='red', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T14:13:25.069858Z","iopub.execute_input":"2023-05-05T14:13:25.070489Z","iopub.status.idle":"2023-05-05T14:13:25.359242Z","shell.execute_reply.started":"2023-05-05T14:13:25.070430Z","shell.execute_reply":"2023-05-05T14:13:25.357723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(18):\n    # Compute F1 score per q\n    m = f1_score(true[1].values, (oof[i].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{i}: F1 score = {m:.3f}')\n    \n# Compute overall F1 score\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('Overall F1 score = {:.3f}'.format(m))","metadata":{"papermill":{"duration":0.771134,"end_time":"2023-02-07T01:02:44.378465","exception":false,"start_time":"2023-02-07T01:02:43.607331","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:13:25.360652Z","iopub.execute_input":"2023-05-05T14:13:25.361318Z","iopub.status.idle":"2023-05-05T14:13:25.730945Z","shell.execute_reply.started":"2023-05-05T14:13:25.361282Z","shell.execute_reply":"2023-05-05T14:13:25.729953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test Data","metadata":{"papermill":{"duration":0.011075,"end_time":"2023-02-07T01:02:44.400918","exception":false,"start_time":"2023-02-07T01:02:44.389843","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel data_label, df, oof, true\n_ = gc.collect()","metadata":{"papermill":{"duration":0.052132,"end_time":"2023-02-07T01:02:44.464739","exception":false,"start_time":"2023-02-07T01:02:44.412607","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:13:25.732159Z","iopub.execute_input":"2023-05-05T14:13:25.733115Z","iopub.status.idle":"2023-05-05T14:13:25.895734Z","shell.execute_reply.started":"2023-05-05T14:13:25.733073Z","shell.execute_reply":"2023-05-05T14:13:25.894567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = fitur_enginer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"papermill":{"duration":1.002014,"end_time":"2023-02-07T01:02:45.47927","exception":false,"start_time":"2023-02-07T01:02:44.477256","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:13:25.897134Z","iopub.execute_input":"2023-05-05T14:13:25.898268Z","iopub.status.idle":"2023-05-05T14:13:28.368251Z","shell.execute_reply.started":"2023-05-05T14:13:25.898225Z","shell.execute_reply":"2023-05-05T14:13:28.367246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.011427,"end_time":"2023-02-07T01:02:45.502331","exception":false,"start_time":"2023-02-07T01:02:45.490904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":0.027432,"end_time":"2023-02-07T01:02:45.541022","exception":false,"start_time":"2023-02-07T01:02:45.51359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:13:28.369559Z","iopub.execute_input":"2023-05-05T14:13:28.370139Z","iopub.status.idle":"2023-05-05T14:13:28.389301Z","shell.execute_reply.started":"2023-05-05T14:13:28.370103Z","shell.execute_reply":"2023-05-05T14:13:28.388323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"papermill":{"duration":0.020233,"end_time":"2023-02-07T01:02:45.57314","exception":false,"start_time":"2023-02-07T01:02:45.552907","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T14:13:28.390723Z","iopub.execute_input":"2023-05-05T14:13:28.391946Z","iopub.status.idle":"2023-05-05T14:13:28.398304Z","shell.execute_reply.started":"2023-05-05T14:13:28.391904Z","shell.execute_reply":"2023-05-05T14:13:28.397014Z"},"trusted":true},"execution_count":null,"outputs":[]}]}