{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Feature Engineering and XGBoost Training Using Only \"fqid\" Feature\nResearch on \"text\" feature can be found [here](http://www.kaggle.com/code/tsuyoshifujii/xgboost-using-only-text-which-text-is-important).\n\nThis code is based on the following great notebooks.\n\n- https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680\n\n- https://www.kaggle.com/code/shashwatraman/gpu-xgb-baseline-using-rapids-cudf-train\n\n- https://www.kaggle.com/code/pourchot/simple-xgb\n\nThe idea of training on GPU and Inference on CPU was based on the following discussion.\n\n- https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/386218","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.metrics import f1_score\nfrom xgboost import XGBClassifier\n\nfrom tqdm.notebook import tqdm\nfrom collections import defaultdict\nfrom itertools import combinations\nimport pickle\nimport warnings\nimport gc\n\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:08:46.332883Z","iopub.execute_input":"2023-04-04T04:08:46.333747Z","iopub.status.idle":"2023-04-04T04:08:47.678034Z","shell.execute_reply.started":"2023-04-04T04:08:46.333709Z","shell.execute_reply":"2023-04-04T04:08:47.676649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Train Data and Labels","metadata":{}},{"cell_type":"code","source":"# READ USER ID ONLY\ntmp = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')\nprint(f'all users: {len(tmp)}')\n\n# COMPUTE READS AND SKIPS\nPIECES = 10\nCHUNK = int(np.ceil(len(tmp)/PIECES))\n\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k * CHUNK\n    b = (k + 1) * CHUNK\n    if b > len(tmp):\n        b = len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1] + r)\n    \nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:08:47.685556Z","iopub.execute_input":"2023-04-04T04:08:47.688502Z","iopub.status.idle":"2023-04-04T04:10:11.175063Z","shell.execute_reply.started":"2023-04-04T04:08:47.688454Z","shell.execute_reply":"2023-04-04T04:10:11.172816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[0])\nprint('Train size of first piece: ', train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:10:11.176831Z","iopub.execute_input":"2023-04-04T04:10:11.177191Z","iopub.status.idle":"2023-04-04T04:10:16.270176Z","shell.execute_reply.started":"2023-04-04T04:10:11.177154Z","shell.execute_reply":"2023-04-04T04:10:16.268979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets['session_id'].apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets['session_id'].apply(lambda x: int(x.split('_')[-1][1:]))\nprint(targets.shape)\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:10:16.273743Z","iopub.execute_input":"2023-04-04T04:10:16.274433Z","iopub.status.idle":"2023-04-04T04:10:17.250465Z","shell.execute_reply.started":"2023-04-04T04:10:16.274398Z","shell.execute_reply":"2023-04-04T04:10:17.249296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:10:17.252150Z","iopub.execute_input":"2023-04-04T04:10:17.253806Z","iopub.status.idle":"2023-04-04T04:10:17.359196Z","shell.execute_reply.started":"2023-04-04T04:10:17.253761Z","shell.execute_reply":"2023-04-04T04:10:17.357877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineer","metadata":{}},{"cell_type":"code","source":"tmp = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', usecols=['text', 'fqid', 'text_fqid'])\nFQIDS = tmp['fqid'].unique()\ndel tmp\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:10:17.360960Z","iopub.execute_input":"2023-04-04T04:10:17.361492Z","iopub.status.idle":"2023-04-04T04:10:46.285887Z","shell.execute_reply.started":"2023-04-04T04:10:17.361447Z","shell.execute_reply":"2023-04-04T04:10:46.284814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(x, grp):\n    # CALCULATE INTERVAL FOR EACH ACTION\n    x['time_diff'] = x['elapsed_time'] - x.groupby('session_id')['elapsed_time'].shift(1)\n    mask = x['time_diff'] < 0\n    x.loc[mask, 'time_diff'] = 0\n    x['time_diff'] = x['time_diff'] / 1000\n    \n    df_final = x.groupby('session_id')['index'].agg('count')\n    df_final.name = 'num_events'\n    df_final = df_final.reset_index()\n    df_final = df_final.set_index('session_id')\n    \n    for c in FQIDS:\n        x[c] = (x['fqid'] == c).astype('int8')\n    for c in FQIDS:\n        df_final[f'{c}_nSum'] = x.groupby('session_id')[c].agg('sum')\n    x.drop(FQIDS, axis=1, inplace=True)\n    \n    for c in FQIDS:\n        df_final[f'{c}_tAve'] = x[x['fqid'] == c].groupby('session_id')['time_diff'].mean()\n        df_final[f'{c}_tStd'] = x[x['fqid'] == c].groupby('session_id')['time_diff'].std()\n        df_final[f'{c}_tSum'] = x[x['fqid'] == c].groupby('session_id')['time_diff'].sum()\n        df_final[f'{c}_tMin'] = x[x['fqid'] == c].groupby('session_id')['time_diff'].min()\n        df_final[f'{c}_tMax'] = x[x['fqid'] == c].groupby('session_id')['time_diff'].max()\n        \n    df_final = df_final.drop(columns='num_events')\n    \n    return df_final","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:10:46.287768Z","iopub.execute_input":"2023-04-04T04:10:46.288176Z","iopub.status.idle":"2023-04-04T04:10:46.300182Z","shell.execute_reply.started":"2023-04-04T04:10:46.288132Z","shell.execute_reply":"2023-04-04T04:10:46.298845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# PROCESS TRAIN DATA IN PIECES\nall_pieces_df1 = []\nall_pieces_df2 = []\nall_pieces_df3 = []\n\nfor k in range(PIECES):\n    SKIPS = 0\n    if k > 0:\n        SKIPS = range(1, skips[k] + 1)\n    train = pd.read_csv(\n        '/kaggle/input/predict-student-performance-from-game-play/train.csv',\n        usecols=['session_id', 'index', 'elapsed_time', 'fqid', 'level_group'],\n        nrows=reads[k],\n        skiprows=SKIPS\n    )\n    train1 = train[train['level_group'] == '0-4']\n    train2 = train[train['level_group'] == '5-12']\n    train3 = train[train['level_group'] == '13-22']\n    \n    df1 = feature_engineer(train1, grp='0-4')\n    df2 = feature_engineer(train2, grp='5-12')\n    df3 = feature_engineer(train3, grp='13-22')\n    \n    all_pieces_df1.append(df1)\n    all_pieces_df2.append(df2)\n    all_pieces_df3.append(df3)\n    \n    print(f'# PIECE {k + 1} COMPLETED')\n    \n# CONCATENATE ALL PIECES\ndel train, train1, train2, train3; gc.collect()\n\ndf1 = pd.concat(all_pieces_df1, axis=0)\ndf2 = pd.concat(all_pieces_df2, axis=0)\ndf3 = pd.concat(all_pieces_df3, axis=0)\nprint(f'df1: {df1.shape}')\nprint(f'df2: {df2.shape}')\nprint(f'df3: {df3.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:10:46.301588Z","iopub.execute_input":"2023-04-04T04:10:46.302634Z","iopub.status.idle":"2023-04-04T04:30:40.783684Z","shell.execute_reply.started":"2023-04-04T04:10:46.302594Z","shell.execute_reply":"2023-04-04T04:30:40.782421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:30:40.786478Z","iopub.execute_input":"2023-04-04T04:30:40.787310Z","iopub.status.idle":"2023-04-04T04:30:40.885142Z","shell.execute_reply.started":"2023-04-04T04:30:40.787262Z","shell.execute_reply":"2023-04-04T04:30:40.884005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null1 = df1.isnull().sum().sort_values(ascending=False) / len(df1)\nnull2 = df2.isnull().sum().sort_values(ascending=False) / len(df2)\nnull3 = df3.isnull().sum().sort_values(ascending=False) / len(df3)\n\ndrop1 = list(null1[null1 > 0.9].index)\ndrop2 = list(null2[null2 > 0.9].index)\ndrop3 = list(null3[null3 > 0.9].index)\n\nprint(f'drop1: {len(drop1)}')\nprint(f'drop2: {len(drop2)}')\nprint(f'drop3: {len(drop3)}')\n\nfor col in tqdm(df1.columns):\n    if df1[col].nunique() == 1:\n        drop1.append(col)\nprint('**********df1 DONE**********')\n\nfor col in tqdm(df2.columns):\n    if df2[col].nunique() == 1:\n        drop2.append(col)\nprint('**********df2 DONE**********')\n\nfor col in tqdm(df3.columns):\n    if df3[col].nunique() == 1:\n        drop3.append(col)\nprint('**********df3 DONE**********')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:30:40.886530Z","iopub.execute_input":"2023-04-04T04:30:40.887540Z","iopub.status.idle":"2023-04-04T04:30:41.968335Z","shell.execute_reply.started":"2023-04-04T04:30:40.887501Z","shell.execute_reply":"2023-04-04T04:30:41.967243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train XGBoost Model","metadata":{}},{"cell_type":"code","source":"FEATURES1 = [c for c in df1.columns if c not in drop1 + ['level_group']]\nFEATURES2 = [c for c in df2.columns if c not in drop2 + ['level_group', 'first_index']]\nFEATURES3 = [c for c in df3.columns if c not in drop3 + ['level_group']]\n\nprint(f'We will train with df1: {len(FEATURES1)}, df2: {len(FEATURES2)}, df3: {len(FEATURES3)} features')\n\nALL_USERS = df1.index.unique()\n\nprint(f'We will train with {len(ALL_USERS)} users info')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:30:41.972593Z","iopub.execute_input":"2023-04-04T04:30:41.973689Z","iopub.status.idle":"2023-04-04T04:30:42.002730Z","shell.execute_reply.started":"2023-04-04T04:30:41.973647Z","shell.execute_reply":"2023-04-04T04:30:42.001290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_base = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\ntrue = oof_base.copy()\noof = oof_base.copy()\nfor k in range(18):\n    tmp = targets.loc[targets['q'] == k + 1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp['correct'].values\n    \ntrue.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:30:42.004364Z","iopub.execute_input":"2023-04-04T04:30:42.004740Z","iopub.status.idle":"2023-04-04T04:30:42.119055Z","shell.execute_reply.started":"2023-04-04T04:30:42.004700Z","shell.execute_reply":"2023-04-04T04:30:42.117926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_fold = 5\ngkf = GroupKFold(n_splits=n_fold)\nbest_iteration_xgb = defaultdict(list)\nfeature_importance_list = []\n\nprint('# START TRAINING')\n\nfor t in range(1, 19):\n    if t <= 3:\n        grp = '0-4'\n        df = df1\n        FEATURES = FEATURES1\n    elif t <= 13:\n        grp = '5-12'\n        df = df2\n        FEATURES = FEATURES2\n    elif t <= 22:\n        grp = '13-22'\n        df = df3\n        FEATURES = FEATURES3\n        \n    print(f'## QUESTION {t} WITH {len(FEATURES)} FEATURES')\n    \n    params = {\n        'booster': 'gbtree',\n        'objective': 'binary:logistic',\n        'tree_method': 'gpu_hist',\n        'eval_metric': 'logloss',\n        'learning_rate': 0.02,\n        'alpha': 8,\n        'max_depth': 4,\n        'n_estimators': 9999,\n        'early_stopping_rounds': 90,\n        'subsample': 0.8,\n        'colsample_bytree': 0.5,\n        'seed': 42\n    }\n    \n    \n    # COMPUTE CV SCORE WITH 5 GROUP K FOLD\n    for i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n        \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets['q'] == t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets['q'] == t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        clf = XGBClassifier(**params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'], eval_set=[(valid_x[FEATURES].astype('float32'), valid_y['correct'])], verbose=0)\n        print(f'### FOLD {i + 1} COMPLETED')\n        best_iteration_xgb[str(t)].append(clf.best_ntree_limit)\n        \n        fold_importance_df = pd.DataFrame()\n        fold_importance_df['feature'] = FEATURES\n        fold_importance_df['importance'] = clf.feature_importances_\n        fold_importance_df['fold'] = i + 1\n        \n        feature_importance_list.append(fold_importance_df)\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:, 1]\n    \nfeature_importance_df = pd.concat(feature_importance_list)\nfeature_importance_df = feature_importance_df.groupby(['feature'])['importance'].agg(['mean']).sort_values(by='mean', ascending=False)\nfeature_importance_df = feature_importance_df.reset_index()\nfeature_importance_df.to_csv('feature_importance_fqid.csv', header=True, index=False)\nfeature_importance_df.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:30:42.120720Z","iopub.execute_input":"2023-04-04T04:30:42.121151Z","iopub.status.idle":"2023-04-04T04:36:21.846045Z","shell.execute_reply.started":"2023-04-04T04:30:42.121110Z","shell.execute_reply":"2023-04-04T04:36:21.844590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Which \"fqid\" Text is Important?","metadata":{}},{"cell_type":"code","source":"# TOP 20% MOST IMPORTANT FEATURES\nn_pick = int(np.ceil(len(feature_importance_df) * 0.2))\nimportant_fqid_tmp = []\nfor i in range(n_pick):\n    tmp = feature_importance_df['feature'][i]\n    fqid = tmp[:-5]\n    important_fqid_tmp.append(fqid)\nimportant_fqid = list(set(important_fqid_tmp))\nprint(len(important_fqid))","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:36:21.847964Z","iopub.execute_input":"2023-04-04T04:36:21.849240Z","iopub.status.idle":"2023-04-04T04:36:21.859514Z","shell.execute_reply.started":"2023-04-04T04:36:21.849165Z","shell.execute_reply":"2023-04-04T04:36:21.858161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHECK THAT EACH TEXT OBTAINED IS INCLUDED IN THE POPULATION \"FQIDS\"\ncounter = 0\nfor tmp_fqid in important_fqid:\n    if tmp_fqid not in FQIDS:\n        counter += 1\nprint(counter)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:36:21.861498Z","iopub.execute_input":"2023-04-04T04:36:21.862371Z","iopub.status.idle":"2023-04-04T04:36:21.873136Z","shell.execute_reply.started":"2023-04-04T04:36:21.862319Z","shell.execute_reply":"2023-04-04T04:36:21.872020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(important_fqid)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:36:21.874811Z","iopub.execute_input":"2023-04-04T04:36:21.876144Z","iopub.status.idle":"2023-04-04T04:36:21.883959Z","shell.execute_reply.started":"2023-04-04T04:36:21.876110Z","shell.execute_reply":"2023-04-04T04:36:21.882826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## An Extra - Evaluate Models","metadata":{}},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []\nthresholds = []\nbest_score = 0\nbest_threshold = 0\n\nfor threshold in np.arange(0.4, 0.81, 0.005):\n    trues = true.values.reshape((-1))\n    preds = (oof.values.reshape((-1)) > threshold).astype('int')\n    f1 = f1_score(trues, preds, average='macro')\n    scores.append(f1)\n    thresholds.append(threshold)\n    \n    if f1 > best_score:\n        best_score = f1\n        best_threshold = threshold\n        \nprint(f'Best score: {best_score}')\nprint(f'Best threshold: {best_threshold}')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:36:21.885725Z","iopub.execute_input":"2023-04-04T04:36:21.886528Z","iopub.status.idle":"2023-04-04T04:36:31.622940Z","shell.execute_reply.started":"2023-04-04T04:36:21.886490Z","shell.execute_reply":"2023-04-04T04:36:31.621604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLDS VS. F1 SCORES\nplt.figure(figsize=(20, 5))\nplt.plot(thresholds, scores, '-o', color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold', size=14)\nplt.ylabel('Validation F1 Score', size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}', size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:36:31.624772Z","iopub.execute_input":"2023-04-04T04:36:31.625220Z","iopub.status.idle":"2023-04-04T04:36:32.003970Z","shell.execute_reply.started":"2023-04-04T04:36:31.625163Z","shell.execute_reply":"2023-04-04T04:36:32.002998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n    # F1 SCORE PER QUESTION\n    true_k = true[k].values\n    oof_k = (oof[k].values > best_threshold).astype('int')\n    f1_k = f1_score(true_k, oof_k, average='macro')\n    \n    print(f'Q{k+1}: F1 = {f1_k:.3f}')\n\n# F1 SCORE OVERALL\ntrue_all = true.values.reshape((-1))\noof_all = (oof.values.reshape((-1)) > best_threshold).astype(int)\nf1_all = f1_score(true_all, oof_all, average='macro')\n\nprint('-'*30)\nprint(f'Overall F1 = {f1_all:.3f}')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T04:36:32.005136Z","iopub.execute_input":"2023-04-04T04:36:32.006040Z","iopub.status.idle":"2023-04-04T04:36:32.256774Z","shell.execute_reply.started":"2023-04-04T04:36:32.005999Z","shell.execute_reply":"2023-04-04T04:36:32.255266Z"},"trusted":true},"execution_count":null,"outputs":[]}]}