{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:11:24.869138Z","iopub.execute_input":"2023-05-05T09:11:24.870428Z","iopub.status.idle":"2023-05-05T09:11:24.876484Z","shell.execute_reply.started":"2023-05-05T09:11:24.870372Z","shell.execute_reply":"2023-05-05T09:11:24.875115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=['session_id'])\nuser = user.groupby('session_id').session_id.agg('count')\n\npcs = 20\nc = int(np.ceil(len(user)/pcs))\n\nbaca = []\nlewat = [0]\n\nfor i in range(pcs):\n    a = i*c\n    b = (i+1)*c\n    if b > len(user): b = len(user)\n    rows = user.iloc[a:b].sum()\n    baca.append(rows)\n    lewat.append(lewat[-1] + rows)\n    \n\nprint(f' we will read the train.csv file in {pcs} pieces of sizes:')\nprint(baca)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:11:24.883830Z","iopub.execute_input":"2023-05-05T09:11:24.884330Z","iopub.status.idle":"2023-05-05T09:12:08.451140Z","shell.execute_reply.started":"2023-05-05T09:11:24.884292Z","shell.execute_reply":"2023-05-05T09:12:08.450246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',nrows=baca[0])\nprint(tmp.shape)\ntmp.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:12:08.452865Z","iopub.execute_input":"2023-05-05T09:12:08.453419Z","iopub.status.idle":"2023-05-05T09:12:12.828900Z","shell.execute_reply.started":"2023-05-05T09:12:08.453386Z","shell.execute_reply":"2023-05-05T09:12:12.827664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_label = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ndata_label['session'] = data_label.session_id.apply(lambda x: int(x.split('_')[0]))\ndata_label['q'] = data_label.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\n\ndata_label['correct'] = data_label['correct'].astype(bool)\nprint(data_label.shape)\ndata_label.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:12:12.830608Z","iopub.execute_input":"2023-05-05T09:12:12.830948Z","iopub.status.idle":"2023-05-05T09:12:13.941324Z","shell.execute_reply.started":"2023-05-05T09:12:12.830919Z","shell.execute_reply":"2023-05-05T09:12:13.940326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time', 'level', 'page', 'room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\nEVENTS = ['navigate_click', 'person_click', 'cutscene_click', 'object_click',\n          'map_hover', 'notification_click', 'map_click', 'observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:12:13.943786Z","iopub.execute_input":"2023-05-05T09:12:13.944450Z","iopub.status.idle":"2023-05-05T09:12:13.951649Z","shell.execute_reply.started":"2023-05-05T09:12:13.944410Z","shell.execute_reply":"2023-05-05T09:12:13.950280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fitur_enginer(train):\n    # Create an empty list to store DataFrames\n    dfs = []\n    \n\n    for c in CATS:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('nunique')\n        tmporary.name = tmporary.name + '_nunique'\n        dfs.append(tmporary)\n        \n\n    for c in NUMS:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('mean')\n        tmporary.name = tmporary.name + '_mean'\n        dfs.append(tmporary)\n    for c in NUMS:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('std')\n        tmporary.name = tmporary.name + '_std'\n        dfs.append(tmporary)\n        \n\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n        \n\n    for c in EVENTS + ['elapsed_time']:\n        tmporary = train.groupby(['session_id', 'level_group'])[c].agg('sum')\n        tmp.name = tmporary.name + '_sum'\n        dfs.append(tmporary)\n        \n        \n    train = train.drop(EVENTS, axis=1)\n    df = pd.concat(dfs, axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:12:13.953569Z","iopub.execute_input":"2023-05-05T09:12:13.953944Z","iopub.status.idle":"2023-05-05T09:12:13.971830Z","shell.execute_reply.started":"2023-05-05T09:12:13.953913Z","shell.execute_reply":"2023-05-05T09:12:13.970198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Process train data in pieces to avoid memory error\nall_pieces = []\nprint(f'Processing train as {pcs} pieces to avoid memory error... ')\nfor i in range(pcs):\n    print(i, ', ', end='')\n    SKIPS = 0\n    if i > 0:\n        SKIPS = range(1, lewat[i] + 1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=baca[i], skiprows=SKIPS)\n    df = fitur_enginer(train)\n    all_pieces.append(df)\n\n# Concatenate all pieces\nprint('\\n')\ndel train\ngc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:12:13.973534Z","iopub.execute_input":"2023-05-05T09:12:13.973948Z","iopub.status.idle":"2023-05-05T09:21:37.563439Z","shell.execute_reply.started":"2023-05-05T09:12:13.973913Z","shell.execute_reply":"2023-05-05T09:21:37.562495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\n\n# Dapat kan ID unik\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:21:37.564825Z","iopub.execute_input":"2023-05-05T09:21:37.565634Z","iopub.status.idle":"2023-05-05T09:21:37.575647Z","shell.execute_reply.started":"2023-05-05T09:21:37.565596Z","shell.execute_reply":"2023-05-05T09:21:37.574733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install xgboost==1.5.0","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:21:37.576710Z","iopub.execute_input":"2023-05-05T09:21:37.577431Z","iopub.status.idle":"2023-05-05T09:21:37.590224Z","shell.execute_reply.started":"2023-05-05T09:21:37.577399Z","shell.execute_reply":"2023-05-05T09:21:37.589001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\n# Definisikan fitur dan print nilai fitur dan user\nFEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')\n\ngkf = GroupKFold(n_splits=5)\n\n# Inisiasi oof prediksi dataframe dan dictionari model\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# Compute Cv Score With 5 Group K Fold\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False }\n    \n    # Iterasi\n    for t in range(1,19):\n        \n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # Train Data\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = data_label.loc[data_label.q==t].set_index('session').loc[train_users]\n        \n        # Valid Data\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = data_label.loc[data_label.q==t].set_index('session').loc[valid_users]\n        \n        # Train Model\n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[(valid_x[FEATURES].astype('float32'), valid_y['correct'])],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # Save Model, Predict Valid OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:21:37.591860Z","iopub.execute_input":"2023-05-05T09:21:37.592198Z","iopub.status.idle":"2023-05-05T09:24:06.660731Z","shell.execute_reply.started":"2023-05-05T09:21:37.592170Z","shell.execute_reply":"2023-05-05T09:24:06.659777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CV Score And Prediction**","metadata":{}},{"cell_type":"code","source":"true = oof.copy()\nfor i in range(18):\n    tmporary = data_label.loc[data_label.q == i+1].set_index('session').loc[ALL_USERS]\n    true[i] = tmporary.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:24:06.663535Z","iopub.execute_input":"2023-05-05T09:24:06.664135Z","iopub.status.idle":"2023-05-05T09:24:06.802377Z","shell.execute_reply.started":"2023-05-05T09:24:06.664101Z","shell.execute_reply":"2023-05-05T09:24:06.801529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []; thresholds = []\nbest_score = 0\nbest_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:24:06.803592Z","iopub.execute_input":"2023-05-05T09:24:06.804081Z","iopub.status.idle":"2023-05-05T09:24:13.525067Z","shell.execute_reply.started":"2023-05-05T09:24:06.804052Z","shell.execute_reply":"2023-05-05T09:24:13.523957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='green')\nplt.scatter([best_threshold], [best_score], color='red', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:24:13.526550Z","iopub.execute_input":"2023-05-05T09:24:13.527753Z","iopub.status.idle":"2023-05-05T09:24:13.843926Z","shell.execute_reply.started":"2023-05-05T09:24:13.527712Z","shell.execute_reply":"2023-05-05T09:24:13.842972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(18):\n    # Compute F1 score per q\n    m = f1_score(true[1].values, (oof[i].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{i}: F1 score = {m:.3f}')\n    \n# Compute overall F1 score\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('Overall F1 score = {:.3f}'.format(m))","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:24:13.845193Z","iopub.execute_input":"2023-05-05T09:24:13.846307Z","iopub.status.idle":"2023-05-05T09:24:14.174894Z","shell.execute_reply.started":"2023-05-05T09:24:13.846255Z","shell.execute_reply":"2023-05-05T09:24:14.174044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#import gc\n#del data_label, df, oof, true\n#_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T09:24:14.176197Z","iopub.execute_input":"2023-05-05T09:24:14.176856Z","iopub.status.idle":"2023-05-05T09:24:14.180977Z","shell.execute_reply.started":"2023-05-05T09:24:14.176822Z","shell.execute_reply":"2023-05-05T09:24:14.179953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}