{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GridSearchCV\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nfrom catboost import CatBoostClassifier\nfrom sklearn.decomposition import PCA","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-11T10:08:22.789777Z","iopub.execute_input":"2023-04-11T10:08:22.790201Z","iopub.status.idle":"2023-04-11T10:08:22.796715Z","shell.execute_reply.started":"2023-04-11T10:08:22.790165Z","shell.execute_reply":"2023-04-11T10:08:22.795438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\ntmp = target.session_id.str.split(\"_\", expand =True)\ntarget[\"user_id\"] = tmp[0].astype(\"int\")\ntarget[\"q\"] = tmp[1].str.slice(1).astype(\"int\")\ntarget.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:06:59.135548Z","iopub.execute_input":"2023-04-11T10:06:59.136579Z","iopub.status.idle":"2023-04-11T10:07:01.323155Z","shell.execute_reply.started":"2023-04-11T10:06:59.136536Z","shell.execute_reply":"2023-04-11T10:07:01.321903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 將USER十等份，之後批次執行特徵工程","metadata":{}},{"cell_type":"code","source":"tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", usecols = ['session_id'])\nnum_userdata = tmp.groupby(\"session_id\").session_id.agg(\"count\") #count是每個user事件的筆數 / sum會將session的值都相加，不是我們要的\n\nDIVIDE= 10\nEQUAL_USER = np.ceil(len(num_userdata)/DIVIDE).astype(\"int\") #將user人數10等份，PART儲存每一等份該有的user數 \n\nread_row = []\nskip_row =[0]\nfor i in range(DIVIDE):\n    a = i * EQUAL_USER\n    b = (i+1) * EQUAL_USER\n    if b > len(tmp): b = len(tmp)\n    r = num_userdata.iloc[a:b].sum() #從第一位user到經過10等份的最後一位user，加總所有事件筆數\n    read_row.append(r)\n    skip_row.append(skip_row[-1]+r)\n    \nprint(tmp.shape)    \nprint(f\"Read_Row: {read_row}\")\nprint(f\"Skip_Row: {skip_row}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-04-11T01:44:44.895025Z","iopub.execute_input":"2023-04-11T01:44:44.896098Z","iopub.status.idle":"2023-04-11T01:46:11.660490Z","shell.execute_reply.started":"2023-04-11T01:44:44.896034Z","shell.execute_reply":"2023-04-11T01:46:11.659197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del tmp\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T01:19:11.812962Z","iopub.execute_input":"2023-04-11T01:19:11.813338Z","iopub.status.idle":"2023-04-11T01:19:11.991010Z","shell.execute_reply.started":"2023-04-11T01:19:11.813302Z","shell.execute_reply":"2023-04-11T01:19:11.989706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 特徵工程","metadata":{}},{"cell_type":"code","source":"CATS =[\"index\",\"event_name\",\"name\",\"room_fqid\"]\nNUM = [\"elapsed_time\",\"level\"]\nEVENT = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\ndef feature_engineer(data):\n    tmp_df = []\n    for col_name in CATS:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"nunique\") #DataFrame\n        tmp.name = col_name + \"_nunique\"\n        tmp_df.append(tmp)\n\n    for col_name in NUM:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"mean\")\n        tmp.name = col_name + \"_mean\"\n        tmp_df.append(tmp)\n    #將各event name取出成為獨立一欄\n    for e in EVENT:\n        data[e] = (data.event_name == e).astype(\"int\")\n    for col_name in EVENT:\n        tmp = data.groupby([\"session_id\", \"level_group\"])[col_name].agg(\"sum\")\n        tmp.name = col_name + \"_sum\"\n        tmp_df.append(tmp)\n    \n    data.drop(EVENT, axis = 1, inplace =True) # 將上面做的獨立出來的Event欄位刪除\n        \n    # \"elapsed_time\" 單獨計算每個level_group所花的時間   \n    tmp = data.groupby([\"session_id\", \"level_group\"])[\"elapsed_time\"].apply(lambda x: x.max() - x.min())\n    tmp.name = \"playtime\" #此關卡所用的時間\n    tmp_df.append(tmp)\n    \n    df = pd.concat(tmp_df, axis = 1)\n    df = df.fillna(-1) #把NULL都補成1\n    df = df.reset_index()\n    df = df.set_index(\"session_id\")\n    \n    #PCA\n    model = PCA(n_components=1).fit(df.iloc[:,1:])\n    pca_df = model.transform(df.iloc[:,1:])\n    tmp_df = pd.concat([df.reset_index().iloc[:,:2],pd.DataFrame(pca_df)], axis = 1)\n    df = tmp_df.set_index(\"session_id\")\n    \n#     df = df.reset_index() #將sesion_id、level_group 從index拉回df column\n#     df = df.set_index(\"session_id\")\n    df.head()\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-11T01:46:11.662450Z","iopub.execute_input":"2023-04-11T01:46:11.663053Z","iopub.status.idle":"2023-04-11T01:46:11.675495Z","shell.execute_reply.started":"2023-04-11T01:46:11.663003Z","shell.execute_reply":"2023-04-11T01:46:11.674246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 分次讀取玩家資料做特徵工程\n# %%time\n\nall_divide = [] #最後要將all_divide所有df合併\nfor i in range(DIVIDE):\n    SKIP = 0 #當i=0時\n    if i> 0: SKIP = range(1, skip_row[i]+1) #參數skiprows的值需要從1開始算需要跳過的筆數\n    train = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",\n                nrows = read_row[i],\n                skiprows = SKIP,\n                usecols=['session_id','index', 'elapsed_time','event_name','name','level','room_fqid','level_group'])\n    df = feature_engineer(train)\n    all_divide.append(df)\n    \ndf = pd.concat(all_divide, axis = 0)\nprint(df.shape)\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T01:46:11.677007Z","iopub.execute_input":"2023-04-11T01:46:11.677717Z","iopub.status.idle":"2023-04-11T01:51:28.922411Z","shell.execute_reply.started":"2023-04-11T01:46:11.677675Z","shell.execute_reply":"2023-04-11T01:51:28.921066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save_df = df","metadata":{"execution":{"iopub.status.busy":"2023-04-11T01:51:28.924620Z","iopub.execute_input":"2023-04-11T01:51:28.925099Z","iopub.status.idle":"2023-04-11T01:51:28.930885Z","shell.execute_reply.started":"2023-04-11T01:51:28.925063Z","shell.execute_reply":"2023-04-11T01:51:28.929658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 呼叫已做好特徵工程的資料","metadata":{}},{"cell_type":"code","source":"# df = pd.read_csv(\"/kaggle/input/keras-predict-student-performance-from-game-play/feature_eng_train.csv\",index_col=\"session_id\")\n# print(df.shape)\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T11:03:19.877994Z","iopub.execute_input":"2023-04-11T11:03:19.878462Z","iopub.status.idle":"2023-04-11T11:03:19.883382Z","shell.execute_reply.started":"2023-04-11T11:03:19.878423Z","shell.execute_reply":"2023-04-11T11:03:19.882265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#以下將程式碼保存起來 以及下載\n# %save test.py 1-16 #數字代表run從幾個到第幾個\n# %load test.py","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:19:09.210108Z","iopub.execute_input":"2023-04-11T02:19:09.210685Z","iopub.status.idle":"2023-04-11T02:19:09.215222Z","shell.execute_reply.started":"2023-04-11T02:19:09.210636Z","shell.execute_reply":"2023-04-11T02:19:09.214164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 不要標準化，預測效果會變差","metadata":{}},{"cell_type":"code","source":"# from sklearn.preprocessing import StandardScaler\n\n# scaler = StandardScaler().fit(df.iloc[:,1:])\n# df_scaled = pd.DataFrame(scaler.transform(df.iloc[:,1:]), \n#                          columns = df.iloc[:,1:].columns)\n# df_scaled.head(3) ","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:24:38.422422Z","iopub.execute_input":"2023-04-11T02:24:38.422834Z","iopub.status.idle":"2023-04-11T02:24:38.445002Z","shell.execute_reply.started":"2023-04-11T02:24:38.422799Z","shell.execute_reply":"2023-04-11T02:24:38.444004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.concat([df.iloc[:,:1].reset_index(),df_scaled], axis = 1).set_index(\"session_id\")\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:24:47.468274Z","iopub.execute_input":"2023-04-11T02:24:47.468702Z","iopub.status.idle":"2023-04-11T02:24:47.507594Z","shell.execute_reply.started":"2023-04-11T02:24:47.468665Z","shell.execute_reply":"2023-04-11T02:24:47.506066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:24:58.523570Z","iopub.execute_input":"2023-04-11T02:24:58.523996Z","iopub.status.idle":"2023-04-11T02:24:58.537052Z","shell.execute_reply.started":"2023-04-11T02:24:58.523960Z","shell.execute_reply":"2023-04-11T02:24:58.536062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:24:59.561322Z","iopub.execute_input":"2023-04-11T02:24:59.561929Z","iopub.status.idle":"2023-04-11T02:25:00.080164Z","shell.execute_reply.started":"2023-04-11T02:24:59.561866Z","shell.execute_reply":"2023-04-11T02:25:00.078632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train CatBoost Model","metadata":{}},{"cell_type":"code","source":"ALL_USER = df.index.unique()\nFEATURE = df.columns[1:]\nprint(ALL_USER.shape, FEATURE.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:42:08.751133Z","iopub.execute_input":"2023-04-11T10:42:08.752002Z","iopub.status.idle":"2023-04-11T10:42:08.760922Z","shell.execute_reply.started":"2023-04-11T10:42:08.751948Z","shell.execute_reply":"2023-04-11T10:42:08.759858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GroupKFold\ngkf =GroupKFold(n_splits = 5) #Each group will appear exactly once in the test set across all folds\noof = pd.DataFrame(np.zeros((len(ALL_USER),18)), index= ALL_USER,columns=range(1,19)) #row: user, col: q1~q18\nmodel ={}\n\ncat_params = {\n     'iterations': 500,\n     'learning_rate':0.05,\n     'depth': 5,\n     'loss_function':'Logloss',\n     'eval_metric':'F1',\n     'random_seed':42,\n     'early_stopping_rounds':30,\n     'verbose':0,\n     'subsample':0.7,\n     'use_best_model': True\n}\n\nfor i,  (train_index, test_index)  in enumerate(gkf.split(X = df,groups=df.index)): #以session_id作為分組\n    print('#'*25)\n    print('Fold',i+1)\n    print('#'*25)\n    for Q in range(1,19):\n        if Q<=3: grp = '0-4'\n        elif Q<=13: grp = '5-12'\n        elif Q<=22: grp = '13-22'\n            \n        train_x = df.iloc[train_index]\n        train_x = train_x[train_x.level_group == grp][FEATURE].astype(\"float32\")\n        train_user = train_x.index.values\n        train_y = target[target.q == Q].set_index(\"user_id\").loc[train_user].correct\n        \n        val_x = df.iloc[test_index]\n        val_x = val_x[val_x.level_group == grp][FEATURE].astype(\"float32\")\n        val_user = val_x.index.values\n        val_y = target[target.q == Q].set_index(\"user_id\").loc[val_user].correct\n\n        clf = CatBoostClassifier(**cat_params)\n        clf.fit(train_x,train_y,\n                eval_set=[(train_x,train_y),(val_x,val_y)],\n                verbose = 0)\n        model[f\"{grp}_{Q}\"] = clf\n        print(f'{Q} ({clf.get_best_iteration()}), ',end='') # 是看metrics的值 哪個時候最高，不是看loss\n        oof.loc[val_user, Q] = clf.predict(val_x, prediction_type='Probability')[:,1] #[:,0] 是分類為0的機率，[:,1]是分類為1的機率\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:56:18.542744Z","iopub.execute_input":"2023-04-11T10:56:18.543540Z","iopub.status.idle":"2023-04-11T10:56:43.250308Z","shell.execute_reply.started":"2023-04-11T10:56:18.543494Z","shell.execute_reply":"2023-04-11T10:56:43.248482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,3))\nplt.subplot(1,2,1)\nloss = clf.get_evals_result()[\"validation_0\"]['Logloss']\nval_loss = clf.get_evals_result()[\"validation_1\"]['Logloss']\nepoch = range(1,len(loss)+1)\nplt.plot(epoch, loss, label=\"loss\")\nplt.plot(epoch, val_loss, label=\"val_loss\")\nplt.legend()\nplt.xlabel(\"epochs\")\nplt.ylabel(\"loss\")\n\nplt.subplot(1,2,2)\nloss = clf.get_evals_result()[\"validation_0\"]['F1']\nval_loss = clf.get_evals_result()[\"validation_1\"][\"F1\"]\nepoch = range(1,len(loss)+1)\nplt.plot(epoch, loss, label=\"F1\")\nplt.plot(epoch, val_loss, label=\"val_F1\")\nplt.legend()\nplt.xlabel(\"epochs\")\nplt.ylabel(\"F1\")\nplt.show()\nprint(clf.get_best_score())","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:57:09.418867Z","iopub.execute_input":"2023-04-11T10:57:09.419266Z","iopub.status.idle":"2023-04-11T10:57:09.712694Z","shell.execute_reply.started":"2023-04-11T10:57:09.419228Z","shell.execute_reply":"2023-04-11T10:57:09.711850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 获取特征重要性得分\n# feature_importances = clf.get_feature_importance()\n# feature_importances_score = []\n# # 输出各个特征的重要性得分\n# for i, score in enumerate(feature_importances):\n#     feature_importances_score.append(score) \n# FEA_IMP =pd.DataFrame({\"feature\": FEATURE, \"score\":feature_importances_score})\n# FEA_IMP = FEA_IMP.sort_values(by=\"score\", ascending=False)\n\n# print(FEA_IMP.loc[FEA_IMP.score > 0])\n# px.bar(FEA_IMP.loc[FEA_IMP.score > 0],x=\"feature\", y=\"score\")","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:57:25.676592Z","iopub.execute_input":"2023-04-11T10:57:25.677022Z","iopub.status.idle":"2023-04-11T10:57:25.681771Z","shell.execute_reply.started":"2023-04-11T10:57:25.676981Z","shell.execute_reply":"2023-04-11T10:57:25.680908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:26:08.012875Z","iopub.execute_input":"2023-04-11T02:26:08.013236Z","iopub.status.idle":"2023-04-11T02:26:08.494947Z","shell.execute_reply.started":"2023-04-11T02:26:08.013203Z","shell.execute_reply":"2023-04-11T02:26:08.493334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find the best F1 score threshold","metadata":{}},{"cell_type":"code","source":"oof.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:26:08.497358Z","iopub.execute_input":"2023-04-11T02:26:08.498165Z","iopub.status.idle":"2023-04-11T02:26:08.530314Z","shell.execute_reply.started":"2023-04-11T02:26:08.498118Z","shell.execute_reply":"2023-04-11T02:26:08.529140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor Q in range(1,19):\n    # GET TRUE LABELS\n    tmp = target.loc[target.q == Q].set_index('user_id').loc[ALL_USER]\n    true[Q] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:57:39.542040Z","iopub.execute_input":"2023-04-11T10:57:39.542498Z","iopub.status.idle":"2023-04-11T10:57:39.658197Z","shell.execute_reply.started":"2023-04-11T10:57:39.542459Z","shell.execute_reply":"2023-04-11T10:57:39.657165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:26:08.673996Z","iopub.execute_input":"2023-04-11T02:26:08.674418Z","iopub.status.idle":"2023-04-11T02:26:08.690821Z","shell.execute_reply.started":"2023-04-11T02:26:08.674383Z","shell.execute_reply":"2023-04-11T02:26:08.689559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []\nthresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds,average='macro')  #average='macro'\n    scores.append(m.astype(\"float32\"))\n    thresholds.append(threshold.astype(\"float32\"))\n    \nbest_score = max(scores)\nbest_threshold = thresholds[scores.index(max(scores))]","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:57:41.572665Z","iopub.execute_input":"2023-04-11T10:57:41.573059Z","iopub.status.idle":"2023-04-11T10:57:47.292385Z","shell.execute_reply.started":"2023-04-11T10:57:41.573024Z","shell.execute_reply":"2023-04-11T10:57:47.291313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter(best_threshold, best_score, color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:57:47.294229Z","iopub.execute_input":"2023-04-11T10:57:47.297176Z","iopub.status.idle":"2023-04-11T10:57:47.556358Z","shell.execute_reply.started":"2023-04-11T10:57:47.297132Z","shell.execute_reply":"2023-04-11T10:57:47.554583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for Q in range(1,19):\n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[Q].values, (oof[Q].values>best_threshold).astype('int'),average='macro') #average='macro'\n    print(f'Q{Q}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'),average='macro')\nprint('Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T10:57:50.489952Z","iopub.execute_input":"2023-04-11T10:57:50.490398Z","iopub.status.idle":"2023-04-11T10:57:50.817817Z","shell.execute_reply.started":"2023-04-11T10:57:50.490356Z","shell.execute_reply":"2023-04-11T10:57:50.816628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer test data","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:26:15.608303Z","iopub.execute_input":"2023-04-11T02:26:15.608895Z","iopub.status.idle":"2023-04-11T02:26:15.621024Z","shell.execute_reply.started":"2023-04-11T02:26:15.608856Z","shell.execute_reply":"2023-04-11T02:26:15.619459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor i, (test, sample_submission) in enumerate(iter_test):\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n#     print(i)\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = model[f'{grp}_{t}']\n        p = clf.predict(df[FEATURE].astype('float32'), prediction_type='Probability')[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = ( p > best_threshold ).astype(\"int\")\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T02:26:15.622631Z","iopub.execute_input":"2023-04-11T02:26:15.623601Z","iopub.status.idle":"2023-04-11T02:26:15.889650Z","shell.execute_reply.started":"2023-04-11T02:26:15.623378Z","shell.execute_reply":"2023-04-11T02:26:15.888703Z"},"trusted":true},"execution_count":null,"outputs":[]}]}