{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\nfrom itertools import groupby\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning) \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-07T14:07:12.323014Z","iopub.execute_input":"2023-06-07T14:07:12.323964Z","iopub.status.idle":"2023-06-07T14:07:12.359318Z","shell.execute_reply.started":"2023-06-07T14:07:12.323914Z","shell.execute_reply":"2023-06-07T14:07:12.357873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Reading in the data </h2>","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ntrain = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(train.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:07:12.361609Z","iopub.execute_input":"2023-06-07T14:07:12.367265Z","iopub.status.idle":"2023-06-07T14:09:23.569830Z","shell.execute_reply.started":"2023-06-07T14:07:12.367196Z","shell.execute_reply":"2023-06-07T14:09:23.568716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:23.571266Z","iopub.execute_input":"2023-06-07T14:09:23.572243Z","iopub.status.idle":"2023-06-07T14:09:23.618747Z","shell.execute_reply.started":"2023-06-07T14:09:23.572187Z","shell.execute_reply":"2023-06-07T14:09:23.617610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reformating the label table to contain the columns \"session\" and \"q\" separately\nt = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv');\nt['session'] = t.session_id.apply(lambda x: int(x.split('_')[0]))\n\nt['q'] = t.session_id.apply(lambda x: int(x.split('_')[1][1:]))\n\nt= t.drop(\"session_id\", axis=1)\nt","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:23.621757Z","iopub.execute_input":"2023-06-07T14:09:23.622633Z","iopub.status.idle":"2023-06-07T14:09:25.135285Z","shell.execute_reply.started":"2023-06-07T14:09:23.622581Z","shell.execute_reply":"2023-06-07T14:09:25.134210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cleanup_nums = {\"event_name\":  \n#                 {'cutscene_click' : 0, \n#                  'person_click' : 1, \n#                  'navigate_click' : 2,\n#                  'observation_click' : 3, \n#                  'notification_click' : 4, \n#                  'object_click' : 5,\n#                  'object_hover' : 6, \n#                  'map_hover' : 7, \n#                  'map_click' : 8, \n#                  'checkpoint' : 9,\n#                  'notebook_click': 10},\n#                 \"name\": \n#                 {'basic' : 0, \n#                  'undefined' : 1,\n#                  'close': 2, \n#                  'open' : 3, \n#                  'prev' : 4, \n#                  'next' : 5},\n#                 \"room_fqid\" : {\n#                     'tunic.historicalsociety.closet' : 1,\n#                     'tunic.historicalsociety.basement' : 2,\n#                     'tunic.historicalsociety.entry' : 3,\n#                     'tunic.historicalsociety.collection' : 4,\n#                     'tunic.historicalsociety.stacks' : 5,\n#                     'tunic.kohlcenter.halloffame' : 6,\n#                     'tunic.capitol_0.hall' : 7, \n#                     'tunic.historicalsociety.closet_dirty' : 8,\n#                     'tunic.historicalsociety.frontdesk' : 9,\n#                     'tunic.humanecology.frontdesk' : 10, \n#                     'tunic.drycleaner.frontdesk' : 11,\n#                     'tunic.library.frontdesk' : 12, \n#                     'tunic.library.microfiche' : 13,\n#                     'tunic.capitol_1.hall' :14, \n#                     'tunic.historicalsociety.cage': 15,\n#                     'tunic.historicalsociety.collection_flag' : 16, \n#                     'tunic.wildlife.center' : 17,\n#                     'tunic.flaghouse.entry' : 18, \n#                     'tunic.capitol_2.hall' : 19}}\n\n# train.replace(cleanup_nums, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.137115Z","iopub.execute_input":"2023-06-07T14:09:25.137936Z","iopub.status.idle":"2023-06-07T14:09:25.144467Z","shell.execute_reply.started":"2023-06-07T14:09:25.137887Z","shell.execute_reply":"2023-06-07T14:09:25.143458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3> Outlier management </h3>\n\nNote: I have not yet found a way to efficiently remove the outliers, without the RAM overflowing. \n\n<h4> Checkpoint outliers </h4>\n\nSessions that contain the \"checkpoint\" event other than three times did not finish the game properly or played it repeatedly.","metadata":{}},{"cell_type":"markdown","source":"Extracting the number of checkpoints per session, followed by gathering the ids, where there are more or less than three checkpoint events.","metadata":{}},{"cell_type":"code","source":"# # extract the number of checkpoint-events per session and gather the ids, where there is more or less than 3 of them\n# checkpoints_per_session = train[train[\"event_name\"] == \"checkpoint\"].groupby(\"session_id\").agg({\"event_name\": \"count\"}).reset_index(\"session_id\")\n# outlier_ids = checkpoints_per_session[checkpoints_per_session[\"event_name\"] != 3][\"session_id\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.146362Z","iopub.execute_input":"2023-06-07T14:09:25.147165Z","iopub.status.idle":"2023-06-07T14:09:25.162734Z","shell.execute_reply.started":"2023-06-07T14:09:25.147117Z","shell.execute_reply":"2023-06-07T14:09:25.161301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dropping the outlier ids from the train data","metadata":{}},{"cell_type":"code","source":"# index = train.index[train['session_id'].isin(outlier_ids)]\n# train.drop(index=index, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.164704Z","iopub.execute_input":"2023-06-07T14:09:25.165254Z","iopub.status.idle":"2023-06-07T14:09:25.176414Z","shell.execute_reply.started":"2023-06-07T14:09:25.165177Z","shell.execute_reply":"2023-06-07T14:09:25.175307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4> Elapsed time outliers </h4>\n\nIn the following i exclude sessions that took over 2 hours and less than 5 minutes from the data.   First I determine the maximal and mean elapsed time for each session and convert them to minutes","metadata":{}},{"cell_type":"code","source":"# time_per_session = train.groupby(['session_id']).agg({'elapsed_time': ['max', \"mean\"], 'session_id': ['count']}).reset_index(\"session_id\")\n# time_per_session.columns = [\"session_id\",\"max_time\", \"mean_time\", \"event_count\"]\n# time_per_session[\"max_time\"] = time_per_session[\"max_time\"]/60000\n# time_per_session[\"mean_time\"] = time_per_session[\"mean_time\"]/60000\n# time_per_session","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.177896Z","iopub.execute_input":"2023-06-07T14:09:25.178515Z","iopub.status.idle":"2023-06-07T14:09:25.190762Z","shell.execute_reply.started":"2023-06-07T14:09:25.178473Z","shell.execute_reply":"2023-06-07T14:09:25.189702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As with the checkpoint outliers, I extract the ids that i classify as outliers and remove them from the train data.","metadata":{}},{"cell_type":"code","source":"# outlier_ids = time_per_session.loc[(time_per_session[\"max_time\"]>=120) | (time_per_session['max_time']< 5),[\"session_id\"]]\n# index = train.index[train['session_id'].isin(outlier_ids)]\n# train.drop(index=index, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.194900Z","iopub.execute_input":"2023-06-07T14:09:25.195589Z","iopub.status.idle":"2023-06-07T14:09:25.203586Z","shell.execute_reply.started":"2023-06-07T14:09:25.195544Z","shell.execute_reply":"2023-06-07T14:09:25.202607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature, event and level-sets:","metadata":{}},{"cell_type":"code","source":"categorical = ['event_name', 'fqid', 'room_fqid']\nnumerical = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\nevents = ['cutscene_click',\n          'person_click',\n          'navigate_click',\n          'observation_click',\n          'notification_click',\n          'object_click',\n          'object_hover',\n          'map_hover',\n          'map_click',\n          'checkpoint',\n          'notebook_click']\n\nlevels = {\"0-4\" : [0,1,2,3,4],\n         \"5-12\" : [5,6,7,8,9,10,12],\n         \"13-22\": [13,14,15,16,17,18,19,20,21,22]}","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.205023Z","iopub.execute_input":"2023-06-07T14:09:25.205626Z","iopub.status.idle":"2023-06-07T14:09:25.216593Z","shell.execute_reply.started":"2023-06-07T14:09:25.205589Z","shell.execute_reply":"2023-06-07T14:09:25.215589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Feature engineering</h2>\n    \nFor new features I used a variety of numerical statistics, grouped by session_id and level_group. Furthermore, I included some features that are grouped by specific levels of the game. The reason behind that is, that it might be reasonable to gain some more detailed impression of how a player behaves throughout the game, specifically on a finer level than just the broad three level groups that we know of.\n\nThis means that the training set is different, i.e. contains more or less features, depending on which level-group the test data is part of. Therefore I generate three different training data sets further down.","metadata":{}},{"cell_type":"code","source":"def create_new_features(data, level_group):\n    \n    df_list = []\n    \n    data[\"elapsed_time_diff\"] = data[\"elapsed_time\"] - data[\"elapsed_time\"].shift(1)\n    data[\"elapsed_time_diff\"] = data[\"elapsed_time\"].clip(lower=0)\n    \n    # ---- event name \n    # number of unique events per session\n    tmp = data.groupby([\"session_id\"])[\"event_name\"].agg(\"nunique\")\n    tmp.name = tmp.name + '_nunique'\n    df_list.append(tmp)\n    # sum of events per session\n    tmp = data.groupby([\"session_id\"])[\"event_name\"].agg(\"count\")\n    tmp.name = tmp.name + '_count'\n    df_list.append(tmp)\n    \n    # number of events in different types per session\n    for event in events:\n        tmp = data[data[\"event_name\"] == event].groupby([\"session_id\"]).agg({\"event_name\":\"count\"})\n        tmp.columns = [event + \"_count\"]\n        tmp.name = event + '_count'\n        df_list.append(tmp)\n        \n    # number of events per level\n    for level in levels[level_group]:\n        tmp = data[data[\"level\"] == level].groupby([\"session_id\"]).agg({\"event_name\":\"count\"})\n        tmp.columns = [str(level) + \"_event_count\"]\n        tmp.name = str(level) + '_count'\n        df_list.append(tmp)\n        \n        # mean hover duration per level per session\n        tmp = data[data[\"level\"] == level].groupby([\"session_id\"]).agg({\"hover_duration\":\"mean\"})\n        tmp.columns = [str(level) + \"_hover_duration\"]\n        tmp.name = str(level) + 'mean'\n        df_list.append(tmp)\n        \n        # max elapsed time per level per session\n        tmp = data[data[\"level\"] == level].groupby([\"session_id\"]).agg({\"elapsed_time\":\"max\"})\n        tmp.columns = [str(level) + \"_elapsed_time\"]\n        tmp.name = str(level) + '_max'\n        df_list.append(tmp)\n        \n    # elapsed time difference\n    tmp = data.groupby([\"session_id\"])[\"elapsed_time_diff\"].agg(\"mean\")\n    tmp.name = tmp.name + '_mean'\n    df_list.append(tmp)\n    \n    tmp = data.groupby([\"session_id\"])[\"elapsed_time_diff\"].agg(\"max\")\n    tmp.name = tmp.name + '_max'\n    df_list.append(tmp)\n    \n    tmp = data.groupby([\"session_id\"])[\"elapsed_time_diff\"].agg(\"min\")\n    tmp.name = tmp.name + '_min'\n    df_list.append(tmp)\n    \n    # number of unique room_fqids per session\n    tmp = data.groupby([\"session_id\"])[\"room_fqid\"].agg(\"nunique\")\n    tmp.name = tmp.name + '_nunique'\n    df_list.append(tmp)\n        \n    # number of fqids per session\n    tmp = data.groupby([\"session_id\"])[\"fqid\"].agg(\"nunique\")\n    tmp.name = tmp.name + '_nunique'\n    df_list.append(tmp)\n    \n    # finish time per session\n    tmp = data.groupby(['session_id'])[\"elapsed_time\"].agg('max')\n    tmp.name = tmp.name + '_max'\n    df_list.append(tmp)\n    \n    # --- hover duration\n    for aggregation in [\"max\", \"mean\", \"std\"]:\n        tmp = data.groupby(['session_id'])[\"hover_duration\"].agg(aggregation)\n        tmp.name = tmp.name + '_' + aggregation\n        df_list.append(tmp)\n    \n\n    # concatenate dataframes along axis 1/columns\n    df = pd.concat(df_list, axis=1)\n    \n    #TODO: na can be cleaned before or try to determine a value\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.218178Z","iopub.execute_input":"2023-06-07T14:09:25.218824Z","iopub.status.idle":"2023-06-07T14:09:25.244231Z","shell.execute_reply.started":"2023-06-07T14:09:25.218784Z","shell.execute_reply":"2023-06-07T14:09:25.242707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"hover_duration\"] = train[\"hover_duration\"].fillna(0)\n\ndf1 = train[train[\"level_group\"] == \"0-4\"]\ndf2 = train[train[\"level_group\"] == \"5-12\"]\ndf3 = train[train[\"level_group\"] == \"13-22\"]\n\ndf1_train = create_new_features(df1, \"0-4\")\ndf2_train = create_new_features(df2, \"5-12\")\ndf3_train = create_new_features(df3, \"13-22\")\n\nfeatures1 = df1_train.columns\nfeatures2 = df2_train.columns\nfeatures3 = df3_train.columns\nprint(df1_train.shape)\nprint(df2_train.shape)\nprint(df3_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:25.245816Z","iopub.execute_input":"2023-06-07T14:09:25.246718Z","iopub.status.idle":"2023-06-07T14:09:56.882124Z","shell.execute_reply.started":"2023-06-07T14:09:25.246676Z","shell.execute_reply":"2023-06-07T14:09:56.880875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nimport matplotlib.pyplot as plt\nimport time\nfrom sklearn.utils import class_weight\nimport sklearn.utils\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.linear_model import LogisticRegression\n","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:56.883917Z","iopub.execute_input":"2023-06-07T14:09:56.884986Z","iopub.status.idle":"2023-06-07T14:09:58.462123Z","shell.execute_reply.started":"2023-06-07T14:09:56.884927Z","shell.execute_reply":"2023-06-07T14:09:58.460475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_users = df1_train.index.unique()\n# we train a model for each question\nmodels = {}\ngkf = GroupKFold(n_splits=3)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:58.463710Z","iopub.execute_input":"2023-06-07T14:09:58.464220Z","iopub.status.idle":"2023-06-07T14:09:58.470495Z","shell.execute_reply.started":"2023-06-07T14:09:58.464178Z","shell.execute_reply":"2023-06-07T14:09:58.469157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:58.472377Z","iopub.execute_input":"2023-06-07T14:09:58.472801Z","iopub.status.idle":"2023-06-07T14:09:58.513327Z","shell.execute_reply.started":"2023-06-07T14:09:58.472760Z","shell.execute_reply":"2023-06-07T14:09:58.511585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**If true, then xgboost will be used, otherwise Random Forest will be used.**","metadata":{}},{"cell_type":"code","source":"xgboost = True","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:58.515879Z","iopub.execute_input":"2023-06-07T14:09:58.517131Z","iopub.status.idle":"2023-06-07T14:09:58.523173Z","shell.execute_reply.started":"2023-06-07T14:09:58.517054Z","shell.execute_reply":"2023-06-07T14:09:58.521688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 10,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4}","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:58.524765Z","iopub.execute_input":"2023-06-07T14:09:58.525485Z","iopub.status.idle":"2023-06-07T14:09:58.536827Z","shell.execute_reply.started":"2023-06-07T14:09:58.525440Z","shell.execute_reply":"2023-06-07T14:09:58.535294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_users = df1_train.index.unique()\nprint(len(all_users))\n\ngkf = GroupKFold(n_splits=5)\ndf_predictions = pd.DataFrame(data=np.zeros((len(all_users),18)), index=all_users)\n\n# a model for each question\nmodels = {}","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:58.538789Z","iopub.execute_input":"2023-06-07T14:09:58.539335Z","iopub.status.idle":"2023-06-07T14:09:58.551607Z","shell.execute_reply.started":"2023-06-07T14:09:58.539283Z","shell.execute_reply":"2023-06-07T14:09:58.550246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Training level-group 0-4","metadata":{}},{"cell_type":"code","source":"for i, (train_index, test_index) in enumerate(gkf.split(X=df1_train[features1], groups=df1_train[features1].index)):\n    print(\"\\n\\nFold:\", i+1)\n\n    level_group = \"0-4\"\n    features = features1\n    for q in range(1,4):\n\n        df_train_x = df1_train.iloc[train_index]\n        train_users = df_train_x.index.values\n        \n        df_train_y = t.loc[t.q == q].set_index('session').loc[train_users]\n        \n        \n        df_valid_x = df1_train.iloc[test_index]\n        valid_users = df_valid_x.index.values\n        df_valid_y = t.loc[t.q == q].set_index('session').loc[valid_users]\n                \n        # training the model\n        if xgboost:\n            model = XGBClassifier(**xgb_params)\n            model.fit(df_train_x[features].astype('float32'), \n                    df_train_y['correct'],\n                    eval_set=[(df_valid_x[features].astype('float32'), df_valid_y['correct'])],\n                    verbose=0)    \n            df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n            print(f'{q}({model.best_ntree_limit}), ',end='')\n            \n        else:\n            model = RandomForestClassifier(n_estimators=100)\n            model.fit(df_train_x[features].astype(\"float32\"), df_train_y[\"correct\"])\n            df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n       \n\n        models[f'{level_group}_{q}'] = model\n","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:09:58.553590Z","iopub.execute_input":"2023-06-07T14:09:58.554042Z","iopub.status.idle":"2023-06-07T14:10:44.963009Z","shell.execute_reply.started":"2023-06-07T14:09:58.553998Z","shell.execute_reply":"2023-06-07T14:10:44.961864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Training level group 5-12","metadata":{}},{"cell_type":"code","source":"for i, (train_index, test_index) in enumerate(gkf.split(X=df2_train[features2], groups=df2_train[features2].index)):\n    print(\"\\n\\nFold:\", i+1)\n\n    level_group = \"5-12\"\n    features = features2\n    for q in range(4,14):\n        \n        df_train_x = df2_train.iloc[train_index]\n        train_users = df_train_x.index.values\n        df_train_y = t.loc[t.q == q].set_index('session').loc[train_users]\n        \n        \n        df_valid_x = df2_train.iloc[test_index]\n        valid_users = df_valid_x.index.values\n        df_valid_y = t.loc[t.q == q].set_index('session').loc[valid_users]\n                \n        # training the model\n        if xgboost:\n            model = XGBClassifier(**xgb_params)\n            model.fit(df_train_x[features].astype('float32'), \n                    df_train_y['correct'],\n                    eval_set=[(df_valid_x[features].astype('float32'), df_valid_y['correct'])],\n                    verbose=0)    \n            df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n            print(f'{q}({model.best_ntree_limit}), ',end='')\n            \n        else:\n            model = RandomForestClassifier(n_estimators=100)\n            model.fit(df_train_x[features].astype(\"float32\"), df_train_y[\"correct\"])\n            df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n       \n\n        models[f'{level_group}_{q}'] = model","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:10:44.967670Z","iopub.execute_input":"2023-06-07T14:10:44.970734Z","iopub.status.idle":"2023-06-07T14:13:36.138653Z","shell.execute_reply.started":"2023-06-07T14:10:44.970678Z","shell.execute_reply":"2023-06-07T14:13:36.137522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Training level group 13-22","metadata":{}},{"cell_type":"code","source":"for i, (train_index, test_index) in enumerate(gkf.split(X=df3_train[features3], groups=df3_train[features3].index)):\n    print(\"\\n\\nFold:\", i+1)\n\n    level_group = \"13-22\"\n    features = features3\n    for q in range(14,19):\n        \n        df_train_x = df3_train.iloc[train_index]\n        train_users = df_train_x.index.values\n        df_train_y = t.loc[t.q == q].set_index('session').loc[train_users]\n        \n        df_valid_x = df3_train.iloc[test_index]\n        valid_users = df_valid_x.index.values\n        df_valid_y = t.loc[t.q == q].set_index('session').loc[valid_users]\n                \n        # training the model\n        if xgboost:\n            model = XGBClassifier(**xgb_params)\n            model.fit(df_train_x[features].astype('float32'), \n                    df_train_y['correct'],\n                    eval_set=[(df_valid_x[features].astype('float32'), df_valid_y['correct'])],\n                    verbose=0)    \n            df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n            print(f'{q}({model.best_ntree_limit}), ',end='')\n            \n        else:\n            model = RandomForestClassifier(n_estimators=100)\n            model.fit(df_train_x[features].astype(\"float32\"), df_train_y[\"correct\"])\n            df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n       \n        models[f'{level_group}_{q}'] = model\n        df_predictions.loc[valid_users, q-1] = model.predict_proba(df_valid_x[features].astype('float32'))[:,1]\n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:13:36.140467Z","iopub.execute_input":"2023-06-07T14:13:36.141220Z","iopub.status.idle":"2023-06-07T14:15:17.004655Z","shell.execute_reply.started":"2023-06-07T14:13:36.141177Z","shell.execute_reply":"2023-06-07T14:15:17.003364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_predictions","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:17.010433Z","iopub.execute_input":"2023-06-07T14:15:17.011284Z","iopub.status.idle":"2023-06-07T14:15:17.048684Z","shell.execute_reply.started":"2023-06-07T14:15:17.011227Z","shell.execute_reply":"2023-06-07T14:15:17.047373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> converting Probabilities to 0 and 1 predictions","metadata":{}},{"cell_type":"code","source":"df_correct = df_predictions.copy()\nfor q in range(18):\n    tmp = t.loc[t.q == q+1].set_index('session')\n    df_correct[q] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:17.050666Z","iopub.execute_input":"2023-06-07T14:15:17.051082Z","iopub.status.idle":"2023-06-07T14:15:17.104990Z","shell.execute_reply.started":"2023-06-07T14:15:17.051034Z","shell.execute_reply":"2023-06-07T14:15:17.103653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []\nthrs = []\nbest_score = 0\nbest_th = 0\n\nfor th in np.arange(0.35, 0.85, 0.01):\n    print(f'{th:.02f}, ',end='')\n    # shape (11779*18,)\n    preds = (df_predictions.values.reshape(-1) > th).astype('int')\n    f1 = f1_score(df_correct.values.reshape(-1), preds, average='macro')   \n    scores.append(f1)\n    thrs.append(th)\n    \n    if f1 > best_score:\n        best_score = f1\n        best_th = th\n","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:17.110857Z","iopub.execute_input":"2023-06-07T14:15:17.111233Z","iopub.status.idle":"2023-06-07T14:15:25.044827Z","shell.execute_reply.started":"2023-06-07T14:15:17.111195Z","shell.execute_reply":"2023-06-07T14:15:25.043608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(20,5))\nplt.plot(thrs, scores, '-o', color='red')\nplt.scatter([best_th], [best_score], color='red', s=300, alpha=1)\nplt.xlabel('Threshold', size=14)\nplt.ylabel('Validation F1 Score', size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_th:.3}', size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:25.046637Z","iopub.execute_input":"2023-06-07T14:15:25.047008Z","iopub.status.idle":"2023-06-07T14:15:25.345584Z","shell.execute_reply.started":"2023-06-07T14:15:25.046971Z","shell.execute_reply":"2023-06-07T14:15:25.344524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Submission","metadata":{}},{"cell_type":"code","source":"import jo_wilder\ntry:\n    jo_wilder.make_env.__called__ = False\n    env.__called__ = False\n    type(env)._state = type(type(env)._state).__dict__['INIT']\nexcept:\n    pass\n\nenv = jo_wilder.make_env()\niter_test = env.iter_test() ","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:25.346651Z","iopub.execute_input":"2023-06-07T14:15:25.346990Z","iopub.status.idle":"2023-06-07T14:15:25.364394Z","shell.execute_reply.started":"2023-06-07T14:15:25.346956Z","shell.execute_reply":"2023-06-07T14:15:25.363371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"level_groups = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\n#for (sample_submission, test) in iter_test:\nfor (test, sample_submission) in iter_test:\n    \n    level_group = test.level_group.values[0]\n    if level_group == \"0-4\":\n        features = features1\n    elif level_group == \"5-12\":\n        features = features2\n    else:\n        features = features3\n    df_test = create_new_features(test, level_group)\n    q_start, q_end = level_groups[level_group]\n    for q in range(q_start, q_end): # e.g. 1,4 - 4,14 - 14,19\n        # the models are saved like this: models[f'{level_group}_{q}'] = model\n        model = models[f'{level_group}_{q}']\n        # this does not work with the next data version\n        #p = model.predict_proba(df_test[features].astype('float32'))[:,1]\n        p = model.predict_proba(df_test[features].astype('float32'))[0,1]\n        # find the session for question number and set its prediction\n        session_mask = sample_submission.session_id.str.contains(f'q{q}')\n        # this does not work with the next data version\n        sample_submission.loc[session_mask,'correct'] = int(p.item() > best_th)\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:25.367950Z","iopub.execute_input":"2023-06-07T14:15:25.368985Z","iopub.status.idle":"2023-06-07T14:15:26.635243Z","shell.execute_reply.started":"2023-06-07T14:15:25.368941Z","shell.execute_reply":"2023-06-07T14:15:26.634128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.read_csv('submission.csv')\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2023-06-07T14:15:26.636492Z","iopub.execute_input":"2023-06-07T14:15:26.636833Z","iopub.status.idle":"2023-06-07T14:15:26.655629Z","shell.execute_reply.started":"2023-06-07T14:15:26.636799Z","shell.execute_reply":"2023-06-07T14:15:26.654568Z"},"trusted":true},"execution_count":null,"outputs":[]}]}