{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport sys\nimport time\nimport pandas as pd\nimport numpy as np\nfrom collections import Counter\nimport string\nimport warnings\n\nwarnings.filterwarnings('error')\nimport lightgbm as lgb\nfrom sklearn.metrics import f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-05-25T14:54:45.993178Z","iopub.execute_input":"2023-05-25T14:54:45.993559Z","iopub.status.idle":"2023-05-25T14:54:45.999653Z","shell.execute_reply.started":"2023-05-25T14:54:45.993528Z","shell.execute_reply":"2023-05-25T14:54:45.998833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_punctuation(lst):\n    punctuation_count = {}\n    for word in lst:\n        for char in word:\n            if char in string.punctuation:\n                if char in punctuation_count:\n                    punctuation_count[char] += 1\n                else:\n                    punctuation_count[char] = 1\n    return punctuation_count\n","metadata":{"execution":{"iopub.status.busy":"2023-05-25T14:54:48.313089Z","iopub.execute_input":"2023-05-25T14:54:48.313479Z","iopub.status.idle":"2023-05-25T14:54:48.319850Z","shell.execute_reply.started":"2023-05-25T14:54:48.313448Z","shell.execute_reply":"2023-05-25T14:54:48.318798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(data=''):\n   \n    user_data = []  # 用来存当前等级处理好的特征数据\n    data = data.groupby(['session_id','level'],as_index=False)\n    def do_data(data):\n\n        data['elapsed_time'] = max(data['elapsed_time'])\n\n\n        event_name_list = ['cutscene_click', 'person_click', 'navigate_click',\n                           'observation_click', 'notification_click', 'object_click',\n                           'object_hover', 'map_hover', 'map_click', 'checkpoint',\n                           'notebook_click']\n\n        event_name_num = Counter(data['event_name'])\n\n        for event_name_list_of_one in event_name_list:\n            if event_name_list_of_one in event_name_num.keys():\n                data[event_name_list_of_one] = event_name_num[event_name_list_of_one]\n            else:\n                data[event_name_list_of_one] = 0\n\n        hover_duration_time_data = data[['event_name', 'hover_duration']]\n                # print(hover_duration_time_data)\n        for hover_duration_time_colname in ['object_hover', 'map_hover']:\n            hover_time_tmpdf = hover_duration_time_data[hover_duration_time_data['event_name'] == hover_duration_time_colname]\n            hover_time = sum(hover_time_tmpdf['hover_duration'])\n            data[hover_duration_time_colname + \"_time_sum\"] = hover_time\n\n        text = data[\"text\"].to_list()\n        text = str(text)\n        punctuation_dic = count_punctuation(text)\n        for i in ['.','!','?']:\n            if i not in punctuation_dic.keys():\n                data[i] = 0\n            else:\n                data[i] = punctuation_dic[i]\n\n\n\n    data = data.apply(do_data)\n    print('-*'*100)\n    print(data)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-05-25T15:58:16.175048Z","iopub.execute_input":"2023-05-25T15:58:16.175444Z","iopub.status.idle":"2023-05-25T15:58:16.187087Z","shell.execute_reply.started":"2023-05-25T15:58:16.175413Z","shell.execute_reply":"2023-05-25T15:58:16.185957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ncol_name = ['session_id', 'index', 'elapsed_time', 'event_name',\n            'name', 'level', 'page', 'room_coor_x', 'room_coor_y',\n            'screen_coor_x', 'screen_coor_y', 'hover_duration', 'text',\n            'fqid', 'room_fqid', 'text_fqid', 'fullscreen', 'hq', 'music',\n            'level_group']\n\ndtypes = {'session_id': np.int64,\n          'index': np.int32,\n          'elapsed_time': np.int32,\n          'event_name': 'category',\n          'name': 'category',\n          'level': np.uint8,\n          'page': 'category',\n          'room_coor_x': np.float32,\n          'room_coor_y': np.float32,\n          'screen_coor_x': np.float32,\n          'screen_coor_y': np.float32,\n          'hover_duration': np.float32,\n          'text': 'category',\n          'fqid': 'category',\n          'room_fqid': 'category',\n          'text_fqid': 'category',\n          'fullscreen': 'category',\n          'hq': 'category',\n          'music': 'category',\n          'level_group': 'category'}\n\nusecol = ['session_id', 'index', 'elapsed_time', 'event_name',\n          'name', 'level', 'page', 'room_coor_x', 'room_coor_y',\n          'screen_coor_x', 'screen_coor_y', 'hover_duration', 'text',\n          'fqid', 'room_fqid', 'text_fqid', 'fullscreen', 'hq', 'music',\n          'level_group']\n\ndef get_train_data(nrowss = 1000000):\n    skiprows = 0\n    dir = \"/kaggle/input/predict-student-performance-from-game-play/train.csv\"\n    df_of_all_user = pd.DataFrame()\n\n    while max_nrow - skiprows > nrowss:  # 如果剩余的数据还很多\n    #     print(\"现在还在while\")\n        print(\"[\"+\"=\"*int((skiprows/max_nrow)*100)+\">\"+str(int((skiprows/max_nrow)*100)))\n        waiter_to_feature_engineer_data = pd.read_csv(filepath_or_buffer=dir,\n                    skiprows=skiprows, nrows=nrowss,names=col_name,low_memory=False)\n\n\n        if skiprows == 0:\n            # 因为重新设定了列名 数据第一行是列名 所以 去掉第一行\n            waiter_to_feature_engineer_data = waiter_to_feature_engineer_data.iloc[1:, ]\n\n            for col_name_tmp, name_type in dtypes.items():\n                if col_name_tmp in usecol:\n                    waiter_to_feature_engineer_data[col_name_tmp] = waiter_to_feature_engineer_data[col_name_tmp].astype(name_type)\n\n        else:\n            for col_name_tmp, name_type in dtypes.items():\n                if col_name_tmp in usecol:\n                    waiter_to_feature_engineer_data[col_name_tmp] = waiter_to_feature_engineer_data[col_name_tmp].astype(name_type)\n\n\n        # print(\"**\"*40)\n        user_set = set(waiter_to_feature_engineer_data.iloc[:,0])  # 用户列表\n        need_del = waiter_to_feature_engineer_data.iloc[-1:, 0]  # 去掉最后一个\n        user_set.remove(int(need_del))\n    #     print(\"这是用户：\", user_set)\n    #     print(need_del)\n        waiter_to_feature_engineer_data = waiter_to_feature_engineer_data[waiter_to_feature_engineer_data[\"session_id\"].isin(user_set)]\n\n    #     for user_id in tqdm(user_set):\n        for user_id in user_set:\n            one_user_data = feature_engineer(waiter_to_feature_engineer_data[waiter_to_feature_engineer_data['session_id']== user_id])\n            one_user_data = pd.DataFrame(one_user_data)\n            df_of_all_user = pd.concat([df_of_all_user,one_user_data],axis=0)\n\n    #     print(df_of_all_user)\n\n\n\n        if skiprows == 0:\n            skiprows += len(waiter_to_feature_engineer_data)+1\n        else:\n            skiprows += len(waiter_to_feature_engineer_data)\n\n        # print(waiter_to_feature_engineer_data)\n    #     print(\"**\"*40)\n\n        # break\n    else:  # 直接读取后面所有的是数据\n    #     print(\"现在是else\")\n        waiter_to_feature_engineer_data = pd.read_csv(\n            filepath_or_buffer=dir,\n             names=col_name,usecols=usecol,\n        skiprows=skiprows,low_memory=False)\n\n        for col_name_tmp, name_type in dtypes.items():\n            if col_name_tmp in usecol:\n                waiter_to_feature_engineer_data[col_name_tmp] = waiter_to_feature_engineer_data[col_name_tmp].astype(name_type)\n\n        user_set = set(waiter_to_feature_engineer_data.iloc[:, 0])  # 用户列表\n\n    #     print(\"这是用户：\", user_set)\n        waiter_to_feature_engineer_data = waiter_to_feature_engineer_data[\n            waiter_to_feature_engineer_data[\"session_id\"].isin(user_set)]\n\n        for user_id in user_set:\n            one_user_data = feature_engineer(\n                waiter_to_feature_engineer_data[waiter_to_feature_engineer_data['session_id'] == user_id])\n            one_user_data = pd.DataFrame(one_user_data)\n            df_of_all_user = pd.concat([df_of_all_user, one_user_data], axis=0)\n\n    #     print(df_of_all_user)\n    return df_of_all_user","metadata":{"execution":{"iopub.status.busy":"2023-05-25T14:54:58.831642Z","iopub.execute_input":"2023-05-25T14:54:58.832407Z","iopub.status.idle":"2023-05-25T14:54:58.858027Z","shell.execute_reply.started":"2023-05-25T14:54:58.832358Z","shell.execute_reply":"2023-05-25T14:54:58.856899Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndata = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",\n                   usecols=['session_id'])\n\nmax_nrow = data.shape[0]\ndel data\nprint(\"该数据总共有 \",max_nrow,\" 条记录！\")","metadata":{"execution":{"iopub.status.busy":"2023-05-25T15:53:35.561692Z","iopub.execute_input":"2023-05-25T15:53:35.562074Z","iopub.status.idle":"2023-05-25T15:54:38.137056Z","shell.execute_reply.started":"2023-05-25T15:53:35.562042Z","shell.execute_reply":"2023-05-25T15:54:38.135903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf = get_train_data(nrowss = 10000)\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-25T16:00:15.176313Z","iopub.execute_input":"2023-05-25T16:00:15.176733Z","iopub.status.idle":"2023-05-25T16:00:18.874782Z","shell.execute_reply.started":"2023-05-25T16:00:15.176699Z","shell.execute_reply":"2023-05-25T16:00:18.873723Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:23:37.584718Z","iopub.execute_input":"2023-05-24T11:23:37.585528Z","iopub.status.idle":"2023-05-24T11:23:38.691562Z","shell.execute_reply.started":"2023-05-24T11:23:37.585485Z","shell.execute_reply":"2023-05-24T11:23:38.690401Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df = df1.copy()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T13:23:01.510236Z","iopub.execute_input":"2023-05-24T13:23:01.510654Z","iopub.status.idle":"2023-05-24T13:23:01.542555Z","shell.execute_reply.started":"2023-05-24T13:23:01.510623Z","shell.execute_reply":"2023-05-24T13:23:01.541339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =  df[df['leve_num'].isin(list(range(1,19)))]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-24T13:23:05.286831Z","iopub.execute_input":"2023-05-24T13:23:05.287220Z","iopub.status.idle":"2023-05-24T13:23:05.628474Z","shell.execute_reply.started":"2023-05-24T13:23:05.287186Z","shell.execute_reply":"2023-05-24T13:23:05.627392Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in tqdm(range(len(labels))):\n# #     if i %1000 == 0:\n# #         print(\"[\"+\"=\"*int((i/len(labels))*100)+\">\"+str(int((i/len(labels))*100)))\n#     correct = labels.iloc[i,1]\n#     session_id = labels.iloc[i,2]\n#     q_num = labels.iloc[i,3]\n    \n#     df.loc[(df['session_id']==session_id) &(df['leve_num']==q_num),'label'] = correct\n    \n# df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T13:23:29.602883Z","iopub.execute_input":"2023-05-24T13:23:29.603318Z","iopub.status.idle":"2023-05-24T13:47:59.353743Z","shell.execute_reply.started":"2023-05-24T13:23:29.603283Z","shell.execute_reply":"2023-05-24T13:47:59.352791Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:02:10.338891Z","iopub.execute_input":"2023-05-24T14:02:10.339589Z","iopub.status.idle":"2023-05-24T14:02:10.395817Z","shell.execute_reply.started":"2023-05-24T14:02:10.339551Z","shell.execute_reply":"2023-05-24T14:02:10.394926Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\n\n\nevaluation_dict ={}\n\n\n\ndef split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = np.unique(dataset['session_id'])\n    split = int(len(USER_LIST) * (1 - 0.20))\n    print(USER_LIST)\n    return dataset.loc[dataset.session_id.isin(USER_LIST[:split])], dataset.loc[dataset.session_id.isin(USER_LIST[split:])]\n\ntrain_x, valid_x = split_dataset(df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:18:51.501366Z","iopub.execute_input":"2023-05-24T14:18:51.501778Z","iopub.status.idle":"2023-05-24T14:18:51.600600Z","shell.execute_reply.started":"2023-05-24T14:18:51.501729Z","shell.execute_reply":"2023-05-24T14:18:51.599575Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:19:04.629643Z","iopub.execute_input":"2023-05-24T14:19:04.630645Z","iopub.status.idle":"2023-05-24T14:19:04.655221Z","shell.execute_reply.started":"2023-05-24T14:19:04.630599Z","shell.execute_reply":"2023-05-24T14:19:04.654400Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VALID_USER_LIST = np.unique(valid_x['session_id'])","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:19:10.602490Z","iopub.execute_input":"2023-05-24T14:19:10.602900Z","iopub.status.idle":"2023-05-24T14:19:10.609385Z","shell.execute_reply.started":"2023-05-24T14:19:10.602854Z","shell.execute_reply":"2023-05-24T14:19:10.608576Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nprediction_df","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:19:11.712294Z","iopub.execute_input":"2023-05-24T14:19:11.713247Z","iopub.status.idle":"2023-05-24T14:19:11.748041Z","shell.execute_reply.started":"2023-05-24T14:19:11.713210Z","shell.execute_reply":"2023-05-24T14:19:11.747254Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:19:13.860569Z","iopub.execute_input":"2023-05-24T14:19:13.861292Z","iopub.status.idle":"2023-05-24T14:19:13.885401Z","shell.execute_reply.started":"2023-05-24T14:19:13.861254Z","shell.execute_reply":"2023-05-24T14:19:13.884569Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for q_no in range(1,19):\n\n    \n    print(\"### q_no\", q_no)\n\n    train_df = train_x.loc[train_x.leve_num== q_no]\n    train_users = np.unique(train_x['session_id'])\n    train_df_label = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    train_df_label = train_df_label['correct']\n    \n    X_test_df = valid_x.loc[valid_x.leve_num== q_no]\n    valid_users = np.unique(valid_x['session_id'])\n    X_test_df_label = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n    X_test_df_label = X_test_df_label['correct']\n\n    \n    print(\"-\"*100)\n\n\n    # 将数据转换为 LightGBM 特定的数据集格式\n    train_data = lgb.Dataset(train_df.iloc[:,3:], label=train_df_label)\n\n    # 设置模型参数\n    params = {\n        'objective': 'binary',\n        'metric': ['f1'],\n        'boosting_type': 'gbdt',\n        'num_leaves': 31,\n        'learning_rate': 0.1,\n        'feature_fraction': 0.9\n    }\n\n    # 训练模型\n    num_rounds = 100  # 迭代次数\n    model = lgb.train(params, train_data, num_rounds)\n    \n    models[f\"question_{q_no}_model\"] = model\n    \n    # 在测试集上进行预测\n    y_pred = model.predict(X_test_df.iloc[:,3:])\n    \n    \n    y_pred_binary = [1 if p >= 0.5 else 0 for p in y_pred]\n    print(\"y_pred_binary\",y_pred_binary[0:6])\n    \n    # 计算评估指标\n    f1 = f1_score(X_test_df_label, y_pred_binary)\n    accuracy = accuracy_score(X_test_df_label, y_pred_binary)\n    print(f\"question_{q_no}_F1 score: \", f1)\n#     print(f\"question_{q_no}_Accuracy: \", accuracy)\n#     evaluation_dict[f\"question_{q_no}_Accuracy\"] = accuracy\n#     print(y_pred_binary)\n    prediction_df.loc[valid_users, q_no-1] = y_pred","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:25:52.104410Z","iopub.execute_input":"2023-05-24T14:25:52.104830Z","iopub.status.idle":"2023-05-24T14:25:57.162555Z","shell.execute_reply.started":"2023-05-24T14:25:52.104795Z","shell.execute_reply":"2023-05-24T14:25:57.161653Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 在上面的训练模型中 把所有X-test的用户拼接起来 放在一起 去重 计算 长度\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:48:49.133700Z","iopub.status.idle":"2023-05-24T11:48:49.134383Z","shell.execute_reply.started":"2023-05-24T11:48:49.134165Z","shell.execute_reply":"2023-05-24T11:48:49.134200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    # Get the true labels.\n    tmp = labels.loc[labels.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values #  验证集里面真实的labels\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:26:09.831204Z","iopub.execute_input":"2023-05-24T14:26:09.831592Z","iopub.status.idle":"2023-05-24T14:26:09.928490Z","shell.execute_reply.started":"2023-05-24T14:26:09.831559Z","shell.execute_reply":"2023-05-24T14:26:09.927586Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_score = 0; best_threshold = 0\n\n# Loop through threshold values from 0.4 to 0.8 and select the threshold with \n# the highest `F1 score`.\nfor threshold in np.arange(0.0,0.8,0.01):\n    \n    y_true_label = true_df.values.reshape((-1))\n    y_pred_label = (prediction_df.values.reshape((-1))>threshold).astype('int')\n    \n    \n    f1 = f1_score(y_true_label,y_pred_label)\n    print(threshold,f1)\n    if f1 > max_score:\n        max_score = f1\n        best_threshold = threshold\n        \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:26:21.097713Z","iopub.execute_input":"2023-05-24T14:26:21.098139Z","iopub.status.idle":"2023-05-24T14:26:22.563690Z","shell.execute_reply.started":"2023-05-24T14:26:21.098105Z","shell.execute_reply":"2023-05-24T14:26:22.562850Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\n\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:29:17.289676Z","iopub.execute_input":"2023-05-24T14:29:17.290789Z","iopub.status.idle":"2023-05-24T14:29:17.317893Z","shell.execute_reply.started":"2023-05-24T14:29:17.290727Z","shell.execute_reply":"2023-05-24T14:29:17.316534Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,5), '5-12':(5,13), '13-22':(13,19)}\n\n\nfor test, sample_submission in iter_test:\n    try:\n        pre_df = feature_engineer(test)\n        pre_df = pd.DataFrame(pre_df)\n    #     print(\"这是原来数据：\")\n    #     print(test)\n    #     print(\"这是处理后数据：\")\n    #     print(pre_df)\n        tmp_test = pd.DataFrame(test)\n        grp = tmp_test['level_group'][0] #要放的是原来的数据的组别\n\n\n        a,b = limits[grp]\n        print(\"~\"*1000)\n        print(\"现在预测的组是：\",grp)\n        pre_df = pre_df[pre_df['group']==grp]\n\n        for t in range(a,b):\n            print(\"-\"*88)\n            print(f\"现在预测的是 {tmp_test['session_id'][0]} 第 {t} 个问题！\")\n            model= models[f\"question_{t}_model\"] \n\n\n            t_q_num_df = pre_df[pre_df['leve_num']== t]\n    #         print(\"shape  t_q_num_df: \",t_q_num_df.shape)\n    #         print(\"  t_q_num_df: \",t_q_num_df)\n            predictions = model.predict(t_q_num_df.iloc[:,3:])\n    #         print(\"  predictions\",predictions)\n            mask = sample_submission.session_id.str.contains(f'q{t}')\n    #         print(\"mask的具体数据：\",mask)\n            n_predictions = (predictions > best_threshold).astype(int)\n    #         print(\"nppreflatten  \",n_predictions.flatten())\n            sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n\n        env.predict(sample_submission[['session_id', 'correct']])\n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:26:40.764547Z","iopub.execute_input":"2023-05-24T14:26:40.765651Z","iopub.status.idle":"2023-05-24T14:26:40.789816Z","shell.execute_reply.started":"2023-05-24T14:26:40.765606Z","shell.execute_reply":"2023-05-24T14:26:40.788431Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:48:49.145710Z","iopub.status.idle":"2023-05-24T11:48:49.146384Z","shell.execute_reply.started":"2023-05-24T11:48:49.146176Z","shell.execute_reply":"2023-05-24T11:48:49.146197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}