{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GridSearchCV\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nfrom catboost import CatBoostClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T16:38:47.263625Z","iopub.execute_input":"2023-06-08T16:38:47.263998Z","iopub.status.idle":"2023-06-08T16:38:47.272569Z","shell.execute_reply.started":"2023-06-08T16:38:47.263969Z","shell.execute_reply":"2023-06-08T16:38:47.271730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\ntmp = target.session_id.str.split(\"_\", expand =True)\ntarget[\"user_id\"] = tmp[0].astype(\"int\")\ntarget[\"q\"] = tmp[1].str.slice(1).astype(\"int\")\ntarget.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T16:38:48.050765Z","iopub.execute_input":"2023-06-08T16:38:48.051138Z","iopub.status.idle":"2023-06-08T16:38:50.398092Z","shell.execute_reply.started":"2023-06-08T16:38:48.051111Z","shell.execute_reply":"2023-06-08T16:38:50.396887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", usecols = ['session_id'])\nnum_userdata = tmp.groupby(\"session_id\").session_id.agg(\"count\") \n\nDIVIDE= 10\nEQUAL_USER = np.ceil(len(num_userdata)/DIVIDE).astype(\"int\") \n\nread_row = []\nskip_row =[0]\nfor i in range(DIVIDE):\n    a = i * EQUAL_USER\n    b = (i+1) * EQUAL_USER\n    if b > len(tmp): b = len(tmp)\n    r = num_userdata.iloc[a:b].sum() \n    read_row.append(r)\n    skip_row.append(skip_row[-1]+r)\n    \nprint(tmp.shape)    \nprint(f\"Read_Row: {read_row}\")\nprint(f\"Skip_Row: {skip_row}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-08T16:39:54.996278Z","iopub.execute_input":"2023-06-08T16:39:54.996701Z","iopub.status.idle":"2023-06-08T16:41:20.804068Z","shell.execute_reply.started":"2023-06-08T16:39:54.996669Z","shell.execute_reply":"2023-06-08T16:41:20.803097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del tmp\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T16:41:20.805653Z","iopub.execute_input":"2023-06-08T16:41:20.806178Z","iopub.status.idle":"2023-06-08T16:41:21.001096Z","shell.execute_reply.started":"2023-06-08T16:41:20.806148Z","shell.execute_reply":"2023-06-08T16:41:21.000141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def delt_time_def(df):\n    df.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    df['d_time'] = df['elapsed_time'].diff(1)\n    df['d_time'].fillna(0, inplace=True)\n    df['delt_time'] = df['d_time'].clip(0, 103000)  \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-08T16:42:22.893534Z","iopub.execute_input":"2023-06-08T16:42:22.893940Z","iopub.status.idle":"2023-06-08T16:42:22.900468Z","shell.execute_reply.started":"2023-06-08T16:42:22.893909Z","shell.execute_reply":"2023-06-08T16:42:22.899437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS =[\"index\",\"event_name\",\"name\",\"room_fqid\"]\nNUM = [\"elapsed_time\",\"level\", \"hover_duration\"]\nEVENT = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n\ndef feature_engineer(data):\n    data = delt_time_def(data)\n    \n    tmp_df = []\n    for col_name in CATS:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"nunique\") #DataFrame\n        tmp.name = col_name + \"_nunique\"\n        tmp_df.append(tmp)\n\n    for col_name in NUM:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"mean\")\n        tmp.name = col_name + \"_mean\"\n        tmp_df.append(tmp)\n    #將各event name取出成為獨立一欄\n    \n    for col_name in NUM:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"std\")\n        tmp.name = col_name + \"_std\"\n        tmp_df.append(tmp)\n    \n    for col_name in NUM:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"max\")\n        tmp.name = col_name + \"_max\"\n        tmp_df.append(tmp)\n        \n    for col_name in NUM:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"min\")\n        tmp.name = col_name + \"_min\"\n        tmp_df.append(tmp)\n        \n    for e in EVENT:\n        data[e] = (data.event_name == e).astype(\"int\")\n    for col_name in EVENT:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"sum\")\n        tmp.name = col_name + \"_sum\"\n        tmp_df.append(tmp)\n        \n    #NEW\n    qvant = data.groupby([\"session_id\", \"level_group\"])['d_time'].quantile(q=0.3)\n    qvant.name = 'qvant1_0_3'\n    tmp_df.append(qvant)\n\n    qvant = data.groupby([\"session_id\", \"level_group\"])['d_time'].quantile(q=0.8)\n    qvant.name = 'qvant2_0_8'\n    tmp_df.append(qvant)\n\n    qvant = data.groupby([\"session_id\", \"level_group\"])['d_time'].quantile(q=0.5)\n    qvant.name = 'qvant3_0_5'\n    tmp_df.append(qvant)\n\n    qvant = data.groupby([\"session_id\", \"level_group\"])['d_time'].quantile(q=0.65)\n    qvant.name = 'qvant4_0_65'\n    tmp_df.append(qvant)\n    \n    data.drop(EVENT, axis = 1, inplace =True) # 將上面做的獨立出來的Event欄位刪除\n        \n    # \"elapsed_time\" 單獨計算每個level_group所花的時間   \n    tmp = data.groupby([\"session_id\", \"level_group\"])[\"elapsed_time\"].apply(lambda x: x.max() - x.min())\n    tmp.name = \"playtime\" #此關卡所用的時間\n    tmp_df.append(tmp)\n        \n    df = pd.concat(tmp_df, axis = 1)\n    df = df.fillna(-1) ##?????????????????????\n    df = df.reset_index() #將sesion_id、level_group 從index拉回df column\n    df = df.set_index(\"session_id\")\n    df.head()\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:22:32.222844Z","iopub.execute_input":"2023-06-08T17:22:32.223252Z","iopub.status.idle":"2023-06-08T17:22:32.244712Z","shell.execute_reply.started":"2023-06-08T17:22:32.223224Z","shell.execute_reply":"2023-06-08T17:22:32.243531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building the model","metadata":{}},{"cell_type":"code","source":"\n# %%time\n\nall_divide = [] \nfor i in range(DIVIDE):\n    SKIP = 0 \n    if i> 0: SKIP = range(1, skip_row[i]+1) \n    train = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",\n                nrows = read_row[i],\n                skiprows = SKIP,\n                usecols=['session_id','index', 'elapsed_time','event_name','name','level','room_fqid','level_group', 'hover_duration']\n                       )\n    df = feature_engineer(train)\n    all_divide.append(df)\n    \ndf = pd.concat(all_divide, axis = 0)\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:22:39.339536Z","iopub.execute_input":"2023-06-08T17:22:39.339919Z","iopub.status.idle":"2023-06-08T17:32:06.595774Z","shell.execute_reply.started":"2023-06-08T17:22:39.339890Z","shell.execute_reply":"2023-06-08T17:32:06.594791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_feature(train):\n    train['session'] = train.index\n    train[\"year\"] = train[\"session\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    train[\"month\"] = train[\"session\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    train[\"second\"] = train[\"session\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n    train = train.drop(['session'], axis=1)\n    return train","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:35:04.010184Z","iopub.execute_input":"2023-06-08T17:35:04.010690Z","iopub.status.idle":"2023-06-08T17:35:04.021871Z","shell.execute_reply.started":"2023-06-08T17:35:04.010659Z","shell.execute_reply":"2023-06-08T17:35:04.020242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#adding time features\ndf = time_feature(df)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:35:05.291673Z","iopub.execute_input":"2023-06-08T17:35:05.292324Z","iopub.status.idle":"2023-06-08T17:35:05.846787Z","shell.execute_reply.started":"2023-06-08T17:35:05.292293Z","shell.execute_reply":"2023-06-08T17:35:05.845664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:35:06.552528Z","iopub.execute_input":"2023-06-08T17:35:06.552925Z","iopub.status.idle":"2023-06-08T17:35:06.598207Z","shell.execute_reply.started":"2023-06-08T17:35:06.552894Z","shell.execute_reply":"2023-06-08T17:35:06.597424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:35:07.557175Z","iopub.execute_input":"2023-06-08T17:35:07.558119Z","iopub.status.idle":"2023-06-08T17:35:07.763387Z","shell.execute_reply.started":"2023-06-08T17:35:07.558083Z","shell.execute_reply":"2023-06-08T17:35:07.762070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train CatBoost Model","metadata":{}},{"cell_type":"code","source":"ALL_USER = df.index.unique()\nFEATURE = df.columns[1:]\nprint(ALL_USER.shape, FEATURE.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T17:35:08.934017Z","iopub.execute_input":"2023-06-08T17:35:08.934411Z","iopub.status.idle":"2023-06-08T17:35:08.942160Z","shell.execute_reply.started":"2023-06-08T17:35:08.934381Z","shell.execute_reply":"2023-06-08T17:35:08.941189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GroupKFold\ngkf =GroupKFold(n_splits = 3) #Each group will appear exactly once in the test set across all folds\noof = pd.DataFrame(np.zeros((len(ALL_USER),18)), index= ALL_USER,columns=range(1,19)) #row: user, col: q1~q18\nmodel ={}\ncat_params = {\n        'iterations': 2000,\n        'early_stopping_rounds': 90,\n        'depth': 4, #5,\n        'learning_rate': 0.02,\n        'loss_function': \"Logloss\",\n        'random_seed': 222222,\n        'metric_period': 1,\n        'subsample': 0.8,\n        'colsample_bylevel': 0.4,\n        'verbose': 0,\n        'l2_leaf_reg': 20,\n    }\n\nfor i,  (train_index, test_index)  in enumerate(gkf.split(X = df,groups=df.index)):\n    print('#'*25)\n    print('Fold',i+1)\n    print('#'*25)\n    for Q in range(1,19):\n        if Q<=3: grp = '0-4'\n        elif Q<=13: grp = '5-12'\n        elif Q<=22: grp = '13-22'\n            \n        train_x = df.iloc[train_index]\n        train_x = train_x[train_x.level_group == grp][FEATURE].astype(\"float32\")\n        train_user = train_x.index.values\n        train_y = target[target.q == Q].set_index(\"user_id\").loc[train_user].correct\n        \n        val_x = df.iloc[test_index]\n        val_x = val_x[val_x.level_group == grp][FEATURE].astype(\"float32\")\n        val_user = val_x.index.values\n        val_y = target[target.q == Q].set_index(\"user_id\").loc[val_user].correct\n\n        clf = CatBoostClassifier(**cat_params)\n        clf.fit(train_x,train_y,\n                eval_set=[(train_x,train_y),(val_x,val_y)],\n                verbose = 0)\n        model[f\"{grp}_{Q}\"] = clf\n        print(f'{Q} ({clf.get_best_iteration()}), ',end='')\n        oof.loc[val_user, Q] = clf.predict(val_x, prediction_type='Probability')[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:22:15.485878Z","iopub.execute_input":"2023-06-08T18:22:15.486280Z","iopub.status.idle":"2023-06-08T18:28:14.451599Z","shell.execute_reply.started":"2023-06-08T18:22:15.486248Z","shell.execute_reply":"2023-06-08T18:28:14.450593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,3))\nplt.subplot(1,2,1)\nloss = clf.get_evals_result()[\"validation_0\"]['Logloss']\nval_loss = clf.get_evals_result()[\"validation_1\"]['Logloss']\nepoch = range(1,len(loss)+1)\nplt.plot(epoch, loss, label=\"loss\")\nplt.plot(epoch, val_loss, label=\"val_loss\")\nplt.legend()\nplt.xlabel(\"epochs\")\nplt.ylabel(\"loss\")\n\n#plt.subplot(1,2,2)\n#loss = clf.get_evals_result()[\"validation_0\"]['F1']\n#val_loss = clf.get_evals_result()[\"validation_1\"][\"F1\"]\n#epoch = range(1,len(loss)+1)\n#plt.plot(epoch, loss, label=\"loss\")\n#plt.plot(epoch, val_loss, label=\"val_loss\")\n#plt.legend()\n#plt.xlabel(\"epochs\")\n#plt.ylabel(\"F1\")\n#plt.show()\n#print(clf.get_best_score())","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:56.573413Z","iopub.execute_input":"2023-06-08T18:33:56.573818Z","iopub.status.idle":"2023-06-08T18:33:56.833222Z","shell.execute_reply.started":"2023-06-08T18:33:56.573789Z","shell.execute_reply":"2023-06-08T18:33:56.832098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = clf.get_feature_importance()\nfeature_importances_score = []\n\nfor i, score in enumerate(feature_importances):\n    feature_importances_score.append(score) \nFEA_IMP =pd.DataFrame({\"feature\": FEATURE, \"score\":feature_importances_score})\nFEA_IMP = FEA_IMP.sort_values(by=\"score\", ascending=False)\n\nprint(FEA_IMP.loc[FEA_IMP.score > 0])\npx.bar(FEA_IMP.loc[FEA_IMP.score > 0],x=\"feature\", y=\"score\")","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:33:58.397393Z","iopub.execute_input":"2023-06-08T18:33:58.398025Z","iopub.status.idle":"2023-06-08T18:33:58.484832Z","shell.execute_reply.started":"2023-06-08T18:33:58.397993Z","shell.execute_reply":"2023-06-08T18:33:58.484047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:00.918740Z","iopub.execute_input":"2023-06-08T18:34:00.919526Z","iopub.status.idle":"2023-06-08T18:34:01.105295Z","shell.execute_reply.started":"2023-06-08T18:34:00.919486Z","shell.execute_reply":"2023-06-08T18:34:01.104117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find the best F1 score threshold","metadata":{}},{"cell_type":"code","source":"oof.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:04.746678Z","iopub.execute_input":"2023-06-08T18:34:04.747064Z","iopub.status.idle":"2023-06-08T18:34:04.776139Z","shell.execute_reply.started":"2023-06-08T18:34:04.747035Z","shell.execute_reply":"2023-06-08T18:34:04.775051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor Q in range(1,19):\n    # GET TRUE LABELS\n    tmp = target.loc[target.q == Q].set_index('user_id').loc[ALL_USER]\n    true[Q] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:08.412595Z","iopub.execute_input":"2023-06-08T18:34:08.413243Z","iopub.status.idle":"2023-06-08T18:34:08.535005Z","shell.execute_reply.started":"2023-06-08T18:34:08.413209Z","shell.execute_reply":"2023-06-08T18:34:08.534004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:11.445815Z","iopub.execute_input":"2023-06-08T18:34:11.446448Z","iopub.status.idle":"2023-06-08T18:34:11.471624Z","shell.execute_reply.started":"2023-06-08T18:34:11.446406Z","shell.execute_reply":"2023-06-08T18:34:11.470513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []\nthresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds,average='macro')  #average='macro'\n    scores.append(m.astype(\"float32\"))\n    thresholds.append(threshold.astype(\"float32\"))\n    \nbest_score = max(scores)\nbest_threshold = thresholds[scores.index(max(scores))]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:11.945623Z","iopub.execute_input":"2023-06-08T18:34:11.946010Z","iopub.status.idle":"2023-06-08T18:34:18.391994Z","shell.execute_reply.started":"2023-06-08T18:34:11.945981Z","shell.execute_reply":"2023-06-08T18:34:18.391144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter(best_threshold, best_score, color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:25.031144Z","iopub.execute_input":"2023-06-08T18:34:25.031567Z","iopub.status.idle":"2023-06-08T18:34:25.349152Z","shell.execute_reply.started":"2023-06-08T18:34:25.031533Z","shell.execute_reply":"2023-06-08T18:34:25.348293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for Q in range(1,19):\n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[Q].values, (oof[Q].values>best_threshold).astype('int'),average='macro') #average='macro'\n    print(f'Q{Q}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'),average='macro')\nprint('Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:34:33.769193Z","iopub.execute_input":"2023-06-08T18:34:33.769602Z","iopub.status.idle":"2023-06-08T18:34:34.105346Z","shell.execute_reply.started":"2023-06-08T18:34:33.769570Z","shell.execute_reply":"2023-06-08T18:34:34.104462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer test data","metadata":{}},{"cell_type":"code","source":"import jo_wilder_310","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:36:03.523819Z","iopub.execute_input":"2023-06-08T18:36:03.524240Z","iopub.status.idle":"2023-06-08T18:36:03.558938Z","shell.execute_reply.started":"2023-06-08T18:36:03.524210Z","shell.execute_reply":"2023-06-08T18:36:03.557943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMPORT KAGGLE API\n#import jo_wilder\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:36:04.671297Z","iopub.execute_input":"2023-06-08T18:36:04.671746Z","iopub.status.idle":"2023-06-08T18:36:04.677639Z","shell.execute_reply.started":"2023-06-08T18:36:04.671689Z","shell.execute_reply":"2023-06-08T18:36:04.676764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor i, (test, sample_submission) in enumerate(iter_test):\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    df = time_feature(df)\n    \n    # INFER TEST DATA\n#     print(i)\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = model[f'{grp}_{t}']\n        #p = clf.predict(df[FEATURE].astype('float32'), prediction_type='Probability')[:,1]\n        p = clf.predict_proba(df[FEATURE].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = ( p > best_threshold ).astype(\"int\")\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:36:13.972242Z","iopub.execute_input":"2023-06-08T18:36:13.973515Z","iopub.status.idle":"2023-06-08T18:36:14.959137Z","shell.execute_reply.started":"2023-06-08T18:36:13.973473Z","shell.execute_reply":"2023-06-08T18:36:14.958293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}