{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Threshold Optimization\n\nStarting this competition, I was pretty annoyed by the fact that we can't optimize the CV score by optimizing the per question F1 score due to the macro averaging of the evaluation. This is a try at mitigating that and trying to find at least something better than an overall global threshold.\n\nThe first part of the notebook is just the same as the code from my other notebook [LGBM Regressor Ensemble Pipeline](https://www.kaggle.com/code/mayukh18/lgbm-regressor-ensemble-pipeline), needed in order to generate the prediction scores to work on. The later part of the notebook is the actual code.","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nfrom tqdm import tqdm\nfrom collections import Counter\nimport itertools\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import f1_score\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom lightgbm.sklearn import LGBMRegressor","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntrain_labels = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\n\n\ntargets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\ntargets.head()\n\nCATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']\n\ndef feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n    \n    tmp = train.groupby(['session_id', 'level_group'])['elapsed_time'].agg(['max', 'min', 'count'])\n    tmp['time'] = tmp['max'] - tmp['min']\n    tmp['avg_time'] = tmp['time']/tmp['count']\n    dfs.append(tmp)\n    \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    \n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df\n\ndf = feature_engineer(train)\ndel train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-22T00:59:30.296905Z","iopub.execute_input":"2023-02-22T00:59:30.297455Z","iopub.status.idle":"2023-02-22T01:01:21.844214Z","shell.execute_reply.started":"2023-02-22T00:59:30.297418Z","shell.execute_reply":"2023-02-22T01:01:21.843097Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\n# print('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\n# print('We will train with', len(ALL_USERS) ,'users info')\n\nN_FOLDS = 5\n\ngkf = GroupKFold(n_splits=N_FOLDS)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH N GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n#     print('#'*25)\n#     print('### Fold',i+1)\n#     print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        model = LGBMRegressor(learning_rate=0.027, \\\n                    num_leaves=15, \\\n                    n_estimators=200, \\\n                    min_child_samples=20, \\\n                    boosting_type='gbdt',\n                    subsample_for_bin=1000,\n                    max_depth=-1,\n                    colsample_bytree=0.8)\n        model.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{i}_{grp}_{t}'] = model\n        oof.loc[valid_users, t-1] = model.predict(valid_x[FEATURES])\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:01:21.845647Z","iopub.execute_input":"2023-02-22T01:01:21.846097Z","iopub.status.idle":"2023-02-22T01:02:06.637805Z","shell.execute_reply.started":"2023-02-22T01:01:21.846062Z","shell.execute_reply":"2023-02-22T01:02:06.636951Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The CV scores with global threshold\n\nThis is the score that we obtain using a single threshold for all the questions.","metadata":{}},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values\n\n# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.5,0.71,0.01):\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')\n    print(f'{threshold:.02f}, --> F1 Score: {m:.05f}')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold\n        \nprint(f\"Best score using global threshold = {best_score} at threshold value = {best_threshold}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T02:27:41.180685Z","iopub.execute_input":"2023-02-22T02:27:41.181057Z","iopub.status.idle":"2023-02-22T02:27:42.555441Z","shell.execute_reply.started":"2023-02-22T02:27:41.181026Z","shell.execute_reply":"2023-02-22T02:27:42.554227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finding better thresholds\n\nNow what we intended to do! We intend to find a set of thresholds for each question which will cumulatively lead to a higher CV score. We'll iteratively go through each permutation to find the optimal set and hope the CV is better than the global threshold.\n\nI tried this in many ways, using numpy and forming a dynamic programming kind of recursive search tree was a good solution but it was slow. Which was only enabling me to run the experiment on a subset of the whole data which might not have been accurate over the whole dataset.\n\nHence, pytorch to the rescue! We'll be using torchmetrics' f1 score and using a threshold generator so that we can do the search iteratively.","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchmetrics import F1Score\nfrom copy import copy\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nf1 = F1Score(task='multiclass', average='macro', num_classes=2).to(device)\n\nooft = torch.Tensor(np.array(oof)).to(device)\ntruet = torch.Tensor(np.array(true)).to(device)\nscore = f1(truet.reshape((-1)), ooft.reshape((-1)) > 0.63)\n\nprint(f\"F1 score with torch = {score}, same as the one with sklearn!\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T03:05:28.483467Z","iopub.execute_input":"2023-02-22T03:05:28.483857Z","iopub.status.idle":"2023-02-22T03:05:28.501760Z","shell.execute_reply.started":"2023-02-22T03:05:28.483824Z","shell.execute_reply":"2023-02-22T03:05:28.500719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate all the permutations of 18 thresholds based on the options given\nclass ThresholdGen(Dataset):\n    def __init__(self, thresh_options):\n        self.n = len(thresh_options)\n        self.thresh_options = thresh_options\n        self.d = [self.n**i for i in range(18)]\n        \n    def __getitem__(self, index):\n        val = copy(index)\n        opt_idx = []\n        for i in range(17,-1,-1):\n            opt_idx.append(self.thresh_options[val//self.d[i]])\n            val = val % self.d[i]\n            \n        return torch.Tensor(opt_idx), index\n    \n    def __len__(self):\n        return self.n**18\n    \n\nthreshdata = ThresholdGen([0.62, 0.63])\nprint(\"A little sample of our threshold generator\")\nprint(threshdata[121])","metadata":{"execution":{"iopub.status.busy":"2023-02-22T03:05:31.843359Z","iopub.execute_input":"2023-02-22T03:05:31.843745Z","iopub.status.idle":"2023-02-22T03:05:31.852002Z","shell.execute_reply.started":"2023-02-22T03:05:31.843713Z","shell.execute_reply":"2023-02-22T03:05:31.850777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The search loop! It takes ~22 min in the competition environment but gets done in ~6 min in a normal/gpu notebook. In any case you need this only in the training phase.","metadata":{}},{"cell_type":"code","source":"max_score = 0.0\nbest_index = None\n\npbar = tqdm(threshdata, position=0, leave=True)\nfor t in pbar:\n    predst = ooft > t[0].to(device)\n    score = f1(predst.reshape(-1), truet.reshape(-1))\n    if score > max_score:\n        max_score = score\n        best_index = t[1]\n        pbar.set_description(f\"Best F1 {max_score}\")\n        \nprint(f\"Best F1 score = {max_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T03:06:44.583578Z","iopub.execute_input":"2023-02-22T03:06:44.584302Z","iopub.status.idle":"2023-02-22T03:13:12.249603Z","shell.execute_reply.started":"2023-02-22T03:06:44.584265Z","shell.execute_reply":"2023-02-22T03:13:12.248761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Small improvement but improvement nonetheless!","metadata":{}},{"cell_type":"code","source":"# best thresholds from above\nthresh1 = threshdata[best_index]\nthresh1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The problem with this is that we are only checking 2 options which is very constrained. However from a computational point of view, you can see we are checking 262144 possibilities for just 2 options. With 3 options, the number increases to 387420489 and the above loop takes 150 hours (more in this env)! So we need to devise some other ways to achieve this. I tried running another search with 2 options, this time 0.62 and 0.64.","metadata":{}},{"cell_type":"code","source":"max_score = 0.0\nbest_index = None\n\nthreshdata = ThresholdGen([0.62, 0.64])\n\npbar = tqdm(threshdata, position=0, leave=True)\nfor t in pbar:\n    predst = ooft > t[0].to(device)\n    score = f1(predst.reshape(-1), truet.reshape(-1))\n    if score > max_score:\n        max_score = score\n        best_index = t[1]\n        pbar.set_description(f\"Best F1 {max_score}\")\n        \nprint(f\"Best F1 score = {max_score}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T03:13:54.178526Z","iopub.execute_input":"2023-02-22T03:13:54.179228Z","iopub.status.idle":"2023-02-22T03:20:28.216067Z","shell.execute_reply.started":"2023-02-22T03:13:54.179191Z","shell.execute_reply":"2023-02-22T03:20:28.215084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best thresholds from above\nthresh2 = threshdata[best_index]\nthresh2","metadata":{"execution":{"iopub.status.busy":"2023-02-22T02:18:12.362088Z","iopub.execute_input":"2023-02-22T02:18:12.362448Z","iopub.status.idle":"2023-02-22T02:18:12.371679Z","shell.execute_reply.started":"2023-02-22T02:18:12.362420Z","shell.execute_reply":"2023-02-22T02:18:12.370491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Combining above two results... can be other ways too..","metadata":{}},{"cell_type":"code","source":"th_opt = torch.max(thresh1[0], thresh2[0])\npredst = ooft > th_opt.to(device)\nscore = f1(predst.reshape(-1), truet.reshape(-1)).item()\n\nprint(f\"We get a new best F1 score of {score}!\")\nprint()\nprint(f\"Best threshold set:\\n{th_opt.cpu().numpy()}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T03:31:04.037594Z","iopub.execute_input":"2023-02-22T03:31:04.038020Z","iopub.status.idle":"2023-02-22T03:31:04.051642Z","shell.execute_reply.started":"2023-02-22T03:31:04.037986Z","shell.execute_reply":"2023-02-22T03:31:04.050108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There can be quite a few improvements on it, the very first one is to be able to experiment with more threshold values and do that within a respectable time.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}