{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score, make_scorer\nfrom tqdm import tqdm\nfrom sklearn.model_selection import RandomizedSearchCV\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-30T00:38:35.953293Z","iopub.execute_input":"2023-08-30T00:38:35.9537Z","iopub.status.idle":"2023-08-30T00:38:35.959612Z","shell.execute_reply.started":"2023-08-30T00:38:35.953669Z","shell.execute_reply":"2023-08-30T00:38:35.958431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=[0])","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:38:35.96146Z","iopub.execute_input":"2023-08-30T00:38:35.962022Z","iopub.status.idle":"2023-08-30T00:40:17.645541Z","shell.execute_reply.started":"2023-08-30T00:38:35.961973Z","shell.execute_reply":"2023-08-30T00:40:17.644565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data['session_id'].groupby(data['session_id']).count()\nPIECES = 10\nCHUNK = int( np.ceil(len(data)/PIECES) )\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k*CHUNK\n    b = (k+1)*CHUNK\n    if b>len(data): b=len(data)\n    r = data.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)\nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:40:17.646972Z","iopub.execute_input":"2023-08-30T00:40:17.647623Z","iopub.status.idle":"2023-08-30T00:40:18.191194Z","shell.execute_reply.started":"2023-08-30T00:40:17.647581Z","shell.execute_reply":"2023-08-30T00:40:18.190316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[0])\nprint('Train size of first piece:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:40:18.193439Z","iopub.execute_input":"2023-08-30T00:40:18.194245Z","iopub.status.idle":"2023-08-30T00:40:27.930604Z","shell.execute_reply.started":"2023-08-30T00:40:18.194197Z","shell.execute_reply":"2023-08-30T00:40:27.929849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( labels.shape )\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:40:27.931642Z","iopub.execute_input":"2023-08-30T00:40:27.93246Z","iopub.status.idle":"2023-08-30T00:40:29.456596Z","shell.execute_reply.started":"2023-08-30T00:40:27.932429Z","shell.execute_reply":"2023-08-30T00:40:29.455832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text_fqid', 'text', 'name']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y','hover_duration']\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:40:29.457606Z","iopub.execute_input":"2023-08-30T00:40:29.458432Z","iopub.status.idle":"2023-08-30T00:40:29.463506Z","shell.execute_reply.started":"2023-08-30T00:40:29.458402Z","shell.execute_reply":"2023-08-30T00:40:29.462488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n########categorical###############\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n###############nums###############\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg(lambda x: x.max() - x.min())\n        tmp.name = tmp.name + '_range'\n        dfs.append(tmp)\n    for c in NUMS: \n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('median')\n        tmp.name = tmp.name + '_median'\n        dfs.append(tmp)\n\n##############EVENTS list###################\"\"    \n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8') \n    for c in EVENTS + ['elapsed_time']: \n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum') \n        tmp.name = tmp.name + '_sum' \n        dfs.append(tmp) \n    for c in EVENTS :\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg(lambda x: x.mode().values[0])\n        tmp.name = tmp.name + '_mode'\n        dfs.append(tmp)\n   \n\n    \n    train = train.drop(EVENTS,axis=1)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:40:29.464838Z","iopub.execute_input":"2023-08-30T00:40:29.465125Z","iopub.status.idle":"2023-08-30T00:40:29.48335Z","shell.execute_reply.started":"2023-08-30T00:40:29.4651Z","shell.execute_reply":"2023-08-30T00:40:29.481403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# PROCESS TRAIN DATA IN PIECES\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineer(train)\n    all_pieces.append(df)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train; gc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:40:29.485138Z","iopub.execute_input":"2023-08-30T00:40:29.485858Z","iopub.status.idle":"2023-08-30T00:54:24.707522Z","shell.execute_reply.started":"2023-08-30T00:40:29.485824Z","shell.execute_reply":"2023-08-30T00:54:24.706318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:54:24.711772Z","iopub.execute_input":"2023-08-30T00:54:24.712138Z","iopub.status.idle":"2023-08-30T00:54:24.726803Z","shell.execute_reply.started":"2023-08-30T00:54:24.712106Z","shell.execute_reply":"2023-08-30T00:54:24.725139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25) \n    xgb_params = {\n    'objective' : 'binary:logitraw',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'reg_alpha' : 0.5,\n    'reg_lambda' : 0.5,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4\n    }\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:54:24.73087Z","iopub.execute_input":"2023-08-30T00:54:24.731263Z","iopub.status.idle":"2023-08-30T00:56:17.356415Z","shell.execute_reply.started":"2023-08-30T00:54:24.731217Z","shell.execute_reply":"2023-08-30T00:56:17.355573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:17.360631Z","iopub.execute_input":"2023-08-30T00:56:17.363307Z","iopub.status.idle":"2023-08-30T00:56:17.492066Z","shell.execute_reply.started":"2023-08-30T00:56:17.363266Z","shell.execute_reply":"2023-08-30T00:56:17.491144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:17.495631Z","iopub.execute_input":"2023-08-30T00:56:17.49598Z","iopub.status.idle":"2023-08-30T00:56:25.027309Z","shell.execute_reply.started":"2023-08-30T00:56:17.495949Z","shell.execute_reply":"2023-08-30T00:56:25.025836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:25.029041Z","iopub.execute_input":"2023-08-30T00:56:25.029457Z","iopub.status.idle":"2023-08-30T00:56:25.430599Z","shell.execute_reply.started":"2023-08-30T00:56:25.029422Z","shell.execute_reply":"2023-08-30T00:56:25.429328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:25.432375Z","iopub.execute_input":"2023-08-30T00:56:25.432781Z","iopub.status.idle":"2023-08-30T00:56:25.812247Z","shell.execute_reply.started":"2023-08-30T00:56:25.432747Z","shell.execute_reply":"2023-08-30T00:56:25.811128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:25.813546Z","iopub.execute_input":"2023-08-30T00:56:25.813892Z","iopub.status.idle":"2023-08-30T00:56:26.263856Z","shell.execute_reply.started":"2023-08-30T00:56:25.813864Z","shell.execute_reply":"2023-08-30T00:56:26.26267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:26.264924Z","iopub.status.idle":"2023-08-30T00:56:26.265622Z","shell.execute_reply.started":"2023-08-30T00:56:26.265367Z","shell.execute_reply":"2023-08-30T00:56:26.265388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"execution":{"iopub.status.busy":"2023-08-30T00:56:26.267132Z","iopub.status.idle":"2023-08-30T00:56:26.267704Z","shell.execute_reply.started":"2023-08-30T00:56:26.267501Z","shell.execute_reply":"2023-08-30T00:56:26.267522Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}