{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-18T20:21:25.058214Z","iopub.execute_input":"2023-04-18T20:21:25.058690Z","iopub.status.idle":"2023-04-18T20:21:25.092171Z","shell.execute_reply.started":"2023-04-18T20:21:25.058650Z","shell.execute_reply":"2023-04-18T20:21:25.090929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename = \"/kaggle/input/predict-student-performance-from-game-play/train.csv\"\n\n# Get total number of rows in the file\ntotal_rows = sum(1 for line in open(filename))\n\n# Calculate number of rows to read\nnrows = int(0.5 * total_rows)\n\n# Read the first nrows rows of the file\ntrain = pd.read_csv(filename, nrows=nrows, header=0)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T21:22:35.204662Z","iopub.execute_input":"2023-04-18T21:22:35.205099Z","iopub.status.idle":"2023-04-18T21:23:34.591456Z","shell.execute_reply.started":"2023-04-18T21:22:35.205062Z","shell.execute_reply":"2023-04-18T21:23:34.590004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:58.164694Z","iopub.execute_input":"2023-04-18T20:22:58.165841Z","iopub.status.idle":"2023-04-18T20:22:58.213824Z","shell.execute_reply.started":"2023-04-18T20:22:58.165798Z","shell.execute_reply":"2023-04-18T20:22:58.212727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:58.214986Z","iopub.execute_input":"2023-04-18T20:22:58.215996Z","iopub.status.idle":"2023-04-18T20:22:58.221793Z","shell.execute_reply.started":"2023-04-18T20:22:58.215955Z","shell.execute_reply":"2023-04-18T20:22:58.220727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_labels(targets):\n    labels = targets\n    labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\n    labels['question'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\n    return labels","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:58.224356Z","iopub.execute_input":"2023-04-18T20:22:58.225551Z","iopub.status.idle":"2023-04-18T20:22:58.234930Z","shell.execute_reply.started":"2023-04-18T20:22:58.225508Z","shell.execute_reply":"2023-04-18T20:22:58.233806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets = feature_labels(targets)\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:58.236758Z","iopub.execute_input":"2023-04-18T20:22:58.237604Z","iopub.status.idle":"2023-04-18T20:22:59.387640Z","shell.execute_reply.started":"2023-04-18T20:22:58.237563Z","shell.execute_reply":"2023-04-18T20:22:59.386681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Reference :** https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680?scriptVersionId=123110383","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:59.389160Z","iopub.execute_input":"2023-04-18T20:22:59.389506Z","iopub.status.idle":"2023-04-18T20:22:59.394965Z","shell.execute_reply.started":"2023-04-18T20:22:59.389474Z","shell.execute_reply":"2023-04-18T20:22:59.393895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:59.396815Z","iopub.execute_input":"2023-04-18T20:22:59.397219Z","iopub.status.idle":"2023-04-18T20:22:59.409307Z","shell.execute_reply.started":"2023-04-18T20:22:59.397178Z","shell.execute_reply":"2023-04-18T20:22:59.407949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = feature_engineer(train)\ndf.sort_values(by=['session_id','level_group'], inplace=True,ascending = [True, True])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:22:59.410961Z","iopub.execute_input":"2023-04-18T20:22:59.411310Z","iopub.status.idle":"2023-04-18T20:23:58.791763Z","shell.execute_reply.started":"2023-04-18T20:22:59.411277Z","shell.execute_reply":"2023-04-18T20:23:58.790822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_SESSIONS = df.index.unique()\nprint('We will train with', len(ALL_SESSIONS) ,'sessions info')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.792977Z","iopub.execute_input":"2023-04-18T20:23:58.793653Z","iopub.status.idle":"2023-04-18T20:23:58.802359Z","shell.execute_reply.started":"2023-04-18T20:23:58.793616Z","shell.execute_reply":"2023-04-18T20:23:58.801527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b = df.index\ntype(b)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.803453Z","iopub.execute_input":"2023-04-18T20:23:58.804114Z","iopub.status.idle":"2023-04-18T20:23:58.817376Z","shell.execute_reply.started":"2023-04-18T20:23:58.804080Z","shell.execute_reply":"2023-04-18T20:23:58.816192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.818873Z","iopub.execute_input":"2023-04-18T20:23:58.819383Z","iopub.status.idle":"2023-04-18T20:23:58.831069Z","shell.execute_reply.started":"2023-04-18T20:23:58.819340Z","shell.execute_reply":"2023-04-18T20:23:58.829670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df.copy()\ndf2 = targets.copy()\n# Extract unique session values from df1\nunique_sessions = df1.index.unique()\n\n# Filter df2 based on session values in df1\ndf2_filtered = df2[df2['session'].isin(unique_sessions)]\n\n# Print the resulting dataframe\ndisplay(df2_filtered)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.832790Z","iopub.execute_input":"2023-04-18T20:23:58.833164Z","iopub.status.idle":"2023-04-18T20:23:58.889360Z","shell.execute_reply.started":"2023-04-18T20:23:58.833131Z","shell.execute_reply":"2023-04-18T20:23:58.888181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = df2_filtered","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.894458Z","iopub.execute_input":"2023-04-18T20:23:58.894828Z","iopub.status.idle":"2023-04-18T20:23:58.899537Z","shell.execute_reply.started":"2023-04-18T20:23:58.894794Z","shell.execute_reply":"2023-04-18T20:23:58.898458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train: ',df.shape,', targets: ',targets.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.901221Z","iopub.execute_input":"2023-04-18T20:23:58.902419Z","iopub.status.idle":"2023-04-18T20:23:58.914043Z","shell.execute_reply.started":"2023-04-18T20:23:58.902377Z","shell.execute_reply":"2023-04-18T20:23:58.913020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold, cross_val_score\nfrom xgboost import XGBClassifier\nfrom xgboost import plot_tree\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:23:58.915530Z","iopub.execute_input":"2023-04-18T20:23:58.916121Z","iopub.status.idle":"2023-04-18T20:24:00.245556Z","shell.execute_reply.started":"2023-04-18T20:23:58.916084Z","shell.execute_reply":"2023-04-18T20:24:00.244091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GKF = GroupKFold(n_splits=5)\nOOF = pd.DataFrame(data=np.zeros((len(ALL_SESSIONS),18)), index=ALL_SESSIONS)\nOOF","metadata":{"execution":{"iopub.status.busy":"2023-04-18T21:22:25.769680Z","iopub.execute_input":"2023-04-18T21:22:25.770154Z","iopub.status.idle":"2023-04-18T21:22:25.809142Z","shell.execute_reply.started":"2023-04-18T21:22:25.770116Z","shell.execute_reply":"2023-04-18T21:22:25.808127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nfor i, (train_index, test_index) in enumerate(GKF.split(X=df, groups=df.index)):\n    \n    print('#'*24,'\\n','### Fold',i,'\\n','#'*24)\n    print(f\"  Train: index={train_index}\")\n    print(f\"  Test:  index={test_index}\")\n    \n    xgb_params = {\n        'objective' : 'binary:logistic',\n        'eval_metric':'logloss',\n        'learning_rate': 0.03,\n        'max_depth': 8,\n        'n_estimators': 1000,\n        'early_stopping_rounds': 50,\n        'tree_method':'hist',\n        'subsample':0.8,\n        'colsample_bytree': 0.4,\n        'use_label_encoder' : False\n    }\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for iterate in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if iterate<=3: group = '0-4'\n        elif iterate<=13: group = '5-12'\n        elif iterate<=22: group = '13-22'\n        \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x['level_group'] == group]\n        train_sessions = train_x.index.values\n        train_y = targets.loc[targets['question'] == iterate].set_index('session').loc[train_sessions]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x['level_group'] == group]\n        valid_sessions = valid_x.index.values\n        valid_y = targets.loc[targets['question'] == iterate].set_index('session').loc[valid_sessions]\n\n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{iterate}({clf.best_ntree_limit}), ',end='')\n        \n        # PLOT GRAPH TREE\n#         plot_tree(clf)\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{group}_{iterate}'] = clf\n        OOF.loc[valid_sessions, iterate-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]","metadata":{"execution":{"iopub.status.busy":"2023-04-18T21:22:15.885837Z","iopub.execute_input":"2023-04-18T21:22:15.886240Z","iopub.status.idle":"2023-04-18T21:22:15.956976Z","shell.execute_reply.started":"2023-04-18T21:22:15.886206Z","shell.execute_reply":"2023-04-18T21:22:15.955690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_label = OOF.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets['question'] == k+1].set_index('session').loc[ALL_SESSIONS]\n    binary_label[k] = tmp['correct'].values\nbinary_label","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:46.119262Z","iopub.execute_input":"2023-04-18T20:27:46.119884Z","iopub.status.idle":"2023-04-18T20:27:46.227302Z","shell.execute_reply.started":"2023-04-18T20:27:46.119841Z","shell.execute_reply":"2023-04-18T20:27:46.226124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (OOF.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(binary_label.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:46.228926Z","iopub.execute_input":"2023-04-18T20:27:46.229541Z","iopub.status.idle":"2023-04-18T20:27:49.402587Z","shell.execute_reply.started":"2023-04-18T20:27:46.229502Z","shell.execute_reply":"2023-04-18T20:27:49.401408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.404379Z","iopub.execute_input":"2023-04-18T20:27:49.404730Z","iopub.status.idle":"2023-04-18T20:27:49.676513Z","shell.execute_reply.started":"2023-04-18T20:27:49.404697Z","shell.execute_reply":"2023-04-18T20:27:49.675565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(binary_label[k].values, (OOF[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(binary_label.values.reshape((-1)), (OOF.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.677477Z","iopub.execute_input":"2023-04-18T20:27:49.677774Z","iopub.status.idle":"2023-04-18T20:27:49.843715Z","shell.execute_reply.started":"2023-04-18T20:27:49.677744Z","shell.execute_reply":"2023-04-18T20:27:49.842773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_labels = OOF.copy()\nsubmission_labels","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.844859Z","iopub.execute_input":"2023-04-18T20:27:49.845863Z","iopub.status.idle":"2023-04-18T20:27:49.873245Z","shell.execute_reply.started":"2023-04-18T20:27:49.845824Z","shell.execute_reply":"2023-04-18T20:27:49.872393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = submission_labels.columns","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.874356Z","iopub.execute_input":"2023-04-18T20:27:49.875401Z","iopub.status.idle":"2023-04-18T20:27:49.880485Z","shell.execute_reply.started":"2023-04-18T20:27:49.875365Z","shell.execute_reply":"2023-04-18T20:27:49.879125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold_submission = 0.61\nfor col in columns:\n    submission_labels.loc[(submission_labels[col] >= threshold_submission),col]=1\n    submission_labels.loc[(submission_labels[col] < threshold_submission),col]=0\nsubmission_labels","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.881891Z","iopub.execute_input":"2023-04-18T20:27:49.882419Z","iopub.status.idle":"2023-04-18T20:27:49.942699Z","shell.execute_reply.started":"2023-04-18T20:27:49.882382Z","shell.execute_reply":"2023-04-18T20:27:49.941613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(columns=['session_id','correct','session_level'])","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.944128Z","iopub.execute_input":"2023-04-18T20:27:49.944480Z","iopub.status.idle":"2023-04-18T20:27:49.951020Z","shell.execute_reply.started":"2023-04-18T20:27:49.944435Z","shell.execute_reply":"2023-04-18T20:27:49.950198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create a new DataFrame to store the converted table\ndf_converted = pd.DataFrame(columns=[\"session_id\", \"correct\", \"session_level\"])\n\n# Iterate over the rows of the original DataFrame\nfor i, row in submission_labels.iterrows():    \n    # Set the session level to 0 for now\n    session_level = 0\n    for k in row.index.values:\n        session_id = str(i) + '_q' + str(k)\n        correct_num = row.iloc[k]\n        # Append a new row to the converted DataFrame\n        df_converted = df_converted.append({\n            \"session_id\": session_id,\n            \"correct\": 0,\n            \"session_level\": session_level\n        }, ignore_index=True)\n\n# Print the converted DataFrame\ndf_converted","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:27:49.952503Z","iopub.execute_input":"2023-04-18T20:27:49.953044Z","iopub.status.idle":"2023-04-18T20:50:40.981345Z","shell.execute_reply.started":"2023-04-18T20:27:49.953011Z","shell.execute_reply":"2023-04-18T20:50:40.979940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import random\nimport random","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:40.982859Z","iopub.execute_input":"2023-04-18T20:50:40.983293Z","iopub.status.idle":"2023-04-18T20:50:40.988228Z","shell.execute_reply.started":"2023-04-18T20:50:40.983258Z","shell.execute_reply":"2023-04-18T20:50:40.987382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list to store the rows\nrows = []\n\n# Iterate over the rows of the original DataFrame\nfor i, row in submission_labels.iterrows():    \n    # Set the session level to 0 for now\n    session_level = 0\n    for k in row.index.values:\n        session_id = str(i) + '_q' + str(k)\n        correct_num = row.iloc[k]\n        if k < 3:\n            session_level = random.randint(0, 2)\n        elif k < 12:\n            session_level = random.randint(3, 11)\n        else:\n            session_level = random.randint(12, 18)\n        # Append the row to the list\n        rows.append({\n            \"session_id\": session_id,\n            \"correct\": int(correct_num),\n            \"session_level\": session_level\n        })\n\n# Create the DataFrame from the list of rows\ndf_converted = pd.DataFrame(rows, columns=[\"session_id\", \"correct\", \"session_level\"])\n\n# Print the converted DataFrame\ndf_converted","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:40.989470Z","iopub.execute_input":"2023-04-18T20:50:40.990256Z","iopub.status.idle":"2023-04-18T20:50:44.523953Z","shell.execute_reply.started":"2023-04-18T20:50:40.990219Z","shell.execute_reply":"2023-04-18T20:50:44.523052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_converted.copy()\nsubmission.head()\n# submission.to_csv('submission1.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:44.525177Z","iopub.execute_input":"2023-04-18T20:50:44.526375Z","iopub.status.idle":"2023-04-18T20:50:44.541499Z","shell.execute_reply.started":"2023-04-18T20:50:44.526245Z","shell.execute_reply":"2023-04-18T20:50:44.540288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Reference:**\n* https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/386583\n* https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680?scriptVersionId=123110383\n> The API will deliver two dataframes in this specific order,\n> for every session+level grouping (one group per session for each checkpoint)\n    \n    import jo_wilder\n    env = jo_wilder.make_env()\n    iter_test = env.iter_test()\n\n    counter = 0\n    \n    for (sample_submission, test) in iter_test:\n        if counter == 0:\n            print(sample_submission.head())\n            print(test.head())\n            print(test.shape)\n\n        ## users make predictions here using the test data\n        sample_submission['correct'] = 0\n\n        ## env.predict appends the session+level sample_submission to the overall\n        ## submission\n        env.predict(sample_submission)\n        counter += 1\n","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel targets, df\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:44.542750Z","iopub.execute_input":"2023-04-18T20:50:44.543581Z","iopub.status.idle":"2023-04-18T20:50:44.982990Z","shell.execute_reply.started":"2023-04-18T20:50:44.543545Z","shell.execute_reply":"2023-04-18T20:50:44.981986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:44.984381Z","iopub.execute_input":"2023-04-18T20:50:44.985230Z","iopub.status.idle":"2023-04-18T20:50:45.338994Z","shell.execute_reply.started":"2023-04-18T20:50:44.985193Z","shell.execute_reply":"2023-04-18T20:50:45.337493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:45.340466Z","iopub.execute_input":"2023-04-18T20:50:45.340804Z","iopub.status.idle":"2023-04-18T20:50:45.357641Z","shell.execute_reply.started":"2023-04-18T20:50:45.340772Z","shell.execute_reply":"2023-04-18T20:50:45.356633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# models['0-4_1']","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:45.358958Z","iopub.execute_input":"2023-04-18T20:50:45.359598Z","iopub.status.idle":"2023-04-18T20:50:45.364079Z","shell.execute_reply.started":"2023-04-18T20:50:45.359560Z","shell.execute_reply":"2023-04-18T20:50:45.362990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for level_group in models:\n#     models[level_group].save_model('model'+ level_group +'.json')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T20:50:45.365281Z","iopub.execute_input":"2023-04-18T20:50:45.366133Z","iopub.status.idle":"2023-04-18T20:50:45.377423Z","shell.execute_reply.started":"2023-04-18T20:50:45.366095Z","shell.execute_reply":"2023-04-18T20:50:45.376198Z"},"trusted":true},"execution_count":null,"outputs":[]}]}