{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"},{"sourceId":5675701,"sourceType":"datasetVersion","datasetId":3244175}],"dockerImageVersionId":30458,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import the Required Libraries","metadata":{"id":"zAXHC6-Tn2O5"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport gc\nimport polars as pl","metadata":{"id":"IanlX-Eqn2O5","execution":{"iopub.status.busy":"2023-06-25T22:04:53.039885Z","iopub.execute_input":"2023-06-25T22:04:53.040570Z","iopub.status.idle":"2023-06-25T22:04:53.226433Z","shell.execute_reply.started":"2023-06-25T22:04:53.040530Z","shell.execute_reply":"2023-06-25T22:04:53.225505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:53.228040Z","iopub.execute_input":"2023-06-25T22:04:53.228585Z","iopub.status.idle":"2023-06-25T22:04:53.232411Z","shell.execute_reply.started":"2023-06-25T22:04:53.228551Z","shell.execute_reply":"2023-06-25T22:04:53.231587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats=pd.read_csv(\"/kaggle/input/featur/feature_sort.csv\")\nfeats_sel=feats[feats['kach']>10]\nfeats_sel_1=feats_sel[feats_sel['quest']<=3]\nfeats_sel_2=feats_sel[feats_sel['quest']<=12]\nfeats_sel_3=feats_sel\n\ncols=['kol_col','col1','val1','col2','val2']\nfeats_sel_1=feats_sel_1[cols].drop_duplicates().reset_index(drop=True)\nfeats_sel_2=feats_sel_2[cols].drop_duplicates().reset_index(drop=True)\nfeats_sel_3=feats_sel_3[cols].drop_duplicates().reset_index(drop=True)\n\nprint(len(feats_sel_1))\nprint(len(feats_sel_2))\nprint(len(feats_sel_3))\nfeats_sel_1.head()","metadata":{"id":"_XItl24kn2O7","execution":{"iopub.status.busy":"2023-06-25T22:04:53.233605Z","iopub.execute_input":"2023-06-25T22:04:53.234100Z","iopub.status.idle":"2023-06-25T22:04:53.411160Z","shell.execute_reply.started":"2023-06-25T22:04:53.234067Z","shell.execute_reply":"2023-06-25T22:04:53.410183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['level','page','room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y','delt_time_next']","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:53.413335Z","iopub.execute_input":"2023-06-25T22:04:53.413845Z","iopub.status.idle":"2023-06-25T22:04:53.418137Z","shell.execute_reply.started":"2023-06-25T22:04:53.413811Z","shell.execute_reply":"2023-06-25T22:04:53.417307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train,feats_sel):\n    \n    train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    train['d_time'] = train['elapsed_time'].diff(1)\n    train['d_time'].fillna(0, inplace=True)\n    train['delt_time'] = train['d_time'].clip(0, 103000)\n    train['delt_time_next'] = train['delt_time'].shift(-1)\n\n    new_train = pd.DataFrame(index=train['session_id'].unique())  \n    new_train['session_id'] = new_train.index \n    \n    train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    \n\n    train['d_time'] = train['elapsed_time'].diff(1)\n    train['d_time'].fillna(0, inplace=True)\n    train['delt_time'] = train['d_time'].clip(0, 103000)\n    train['delt_time_next'] = train['delt_time'].shift(-1)\n    \n###\n    base=True\n    if base:\n        for c in CATEGORICAL:\n            new_train[f'{c}_nunique'] = train.groupby(['session_id'])[c].agg('nunique')\n        for c in NUMERICAL:\n            new_train[f'{c}_mean'] = train.groupby(['session_id'])[c].agg('mean')\n            new_train[f'{c}_sum'] = train.groupby(['session_id'])[c].agg('sum')\n###\n    new_train['session_index_count'] = train.groupby(['session_id'])['index'].count()\n    new_train['d_time_q3'] = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    new_train['d_time_q8'] = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    new_train['d_time_q5'] = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    new_train['d_time_q65'] = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    \n    new_train['hover_duration_mean'] = train.groupby(['session_id'])['hover_duration'].agg('mean')\n    new_train['hover_duration_std'] = train.groupby(['session_id'])['hover_duration'].agg('std') \n    new_train['delt_time_mean'] = train.groupby(['session_id'])['delt_time'].agg('mean')\n    new_train['delt_time_std'] = train.groupby(['session_id'])['delt_time'].agg('std') \n    new_train['delt_time_max'] = train.groupby(['session_id'])['delt_time'].agg('max')\n    new_train['delt_time_min'] = train.groupby(['session_id'])['delt_time'].agg('min') \n    \n    new_train['year'] = new_train['session_id'].apply(lambda x: int(str(x)[:2])).astype(np.uint8) # \"year\"\n    new_train['month'] = new_train['session_id'].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8) # \"month\"\n    new_train['day'] = new_train['session_id'].apply(lambda x: int(str(x)[4:6])).astype(np.uint8) # \"day\"\n    new_train['sess_time'] = new_train['session_id'].apply(lambda x: int(str(x)[6:8])).astype(np.uint8) + new_train['session_id'].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)/60\n    new_train = new_train.fillna(-1)\n    \n    t1=feats_sel[feats_sel['kol_col']==1]\n    for i in range(len(t1)):\n        col1 = t1['col1'].iloc[i]\n        val1 = t1['val1'].iloc[i]\n\n        maska1 = (train[col1] == val1)\n        new_train[f'{col1}_{hash(val1)}_delt_time_next_sum'] = train[maska1].groupby(['session_id'])['delt_time_next'].sum()\n        new_train[f'{col1}_{hash(val1)}_delt_time_mean'] = train[maska1].groupby(['session_id'])['delt_time'].mean()\n        new_train[f'{col1}_{hash(val1)}_index_count'] = train[maska1].groupby(['session_id'])['index'].count()\n\n    t2=feats_sel[feats_sel['kol_col']==2]\n    for i in range(len(t2)):\n        col1 = t2['col1'].iloc[i]\n        val1 = t2['val1'].iloc[i]\n        col2 = t2['col2'].iloc[i]\n        val2 = t2['val2'].iloc[i]\n\n        maska2 = (train[col1] == val1) & (train[col2] == val2)\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_next_sum'] = train[maska2].groupby(['session_id'])['delt_time_next'].sum()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_mean'] = train[maska2].groupby(['session_id'])['delt_time'].mean()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_index_count'] = train[maska2].groupby(['session_id'])['index'].count()\n    \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:53.419366Z","iopub.execute_input":"2023-06-25T22:04:53.419860Z","iopub.status.idle":"2023-06-25T22:04:53.444490Z","shell.execute_reply.started":"2023-06-25T22:04:53.419828Z","shell.execute_reply":"2023-06-25T22:04:53.443561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'str',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'\n    }\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes) ###, nrows=500000)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:53.445741Z","iopub.execute_input":"2023-06-25T22:04:53.446298Z","iopub.status.idle":"2023-06-25T22:04:55.464814Z","shell.execute_reply.started":"2023-06-25T22:04:53.446255Z","shell.execute_reply":"2023-06-25T22:04:55.463908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:55.465857Z","iopub.execute_input":"2023-06-25T22:04:55.466375Z","iopub.status.idle":"2023-06-25T22:04:55.568855Z","shell.execute_reply.started":"2023-06-25T22:04:55.466343Z","shell.execute_reply":"2023-06-25T22:04:55.567741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"id":"KD4uayl2n2O9","execution":{"iopub.status.busy":"2023-06-25T22:04:55.570058Z","iopub.execute_input":"2023-06-25T22:04:55.570407Z","iopub.status.idle":"2023-06-25T22:04:55.936493Z","shell.execute_reply.started":"2023-06-25T22:04:55.570373Z","shell.execute_reply":"2023-06-25T22:04:55.935403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"id":"Kva8_Dbqn2O9","execution":{"iopub.status.busy":"2023-06-25T22:04:55.937667Z","iopub.execute_input":"2023-06-25T22:04:55.938697Z","iopub.status.idle":"2023-06-25T22:04:56.689401Z","shell.execute_reply.started":"2023-06-25T22:04:55.938658Z","shell.execute_reply":"2023-06-25T22:04:56.688302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first 5 examples\nlabels.head(5)","metadata":{"id":"0eD-KZMvn2O-","execution":{"iopub.status.busy":"2023-06-25T22:04:56.692989Z","iopub.execute_input":"2023-06-25T22:04:56.693329Z","iopub.status.idle":"2023-06-25T22:04:56.703450Z","shell.execute_reply.started":"2023-06-25T22:04:56.693296Z","shell.execute_reply":"2023-06-25T22:04:56.702599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_1 = dataset_df[dataset_df.level_group=='0-4']\ndataset_df_2 = dataset_df[dataset_df.level_group!='13-22']\ndataset_df_3 = dataset_df\nprint(dataset_df_1.shape)\nprint(dataset_df_2.shape)\nprint(dataset_df_3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:56.704486Z","iopub.execute_input":"2023-06-25T22:04:56.705405Z","iopub.status.idle":"2023-06-25T22:04:56.751082Z","shell.execute_reply.started":"2023-06-25T22:04:56.705362Z","shell.execute_reply":"2023-06-25T22:04:56.750297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del dataset_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:56.752145Z","iopub.execute_input":"2023-06-25T22:04:56.752658Z","iopub.status.idle":"2023-06-25T22:04:56.891820Z","shell.execute_reply.started":"2023-06-25T22:04:56.752626Z","shell.execute_reply":"2023-06-25T22:04:56.889766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_1 = feature_engineer(dataset_df_1,feats_sel_1)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df_1.shape))\ndataset_df_1.head()","metadata":{"id":"JKcoPoemn2PA","execution":{"iopub.status.busy":"2023-06-25T22:04:56.896069Z","iopub.execute_input":"2023-06-25T22:04:56.896520Z","iopub.status.idle":"2023-06-25T22:04:57.581971Z","shell.execute_reply.started":"2023-06-25T22:04:56.896477Z","shell.execute_reply":"2023-06-25T22:04:57.580888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:57.585612Z","iopub.execute_input":"2023-06-25T22:04:57.587588Z","iopub.status.idle":"2023-06-25T22:04:57.693676Z","shell.execute_reply.started":"2023-06-25T22:04:57.587547Z","shell.execute_reply":"2023-06-25T22:04:57.691606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_2 = feature_engineer(dataset_df_2,feats_sel_2)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df_2.shape))\ndataset_df_2.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:04:57.694935Z","iopub.execute_input":"2023-06-25T22:04:57.695684Z","iopub.status.idle":"2023-06-25T22:05:00.057800Z","shell.execute_reply.started":"2023-06-25T22:04:57.695646Z","shell.execute_reply":"2023-06-25T22:05:00.056956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:00.059076Z","iopub.execute_input":"2023-06-25T22:05:00.059599Z","iopub.status.idle":"2023-06-25T22:05:00.161042Z","shell.execute_reply.started":"2023-06-25T22:05:00.059566Z","shell.execute_reply":"2023-06-25T22:05:00.160230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_3 = feature_engineer(dataset_df_3,feats_sel_3)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df_3.shape))\ndataset_df_3.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:00.162272Z","iopub.execute_input":"2023-06-25T22:05:00.162764Z","iopub.status.idle":"2023-06-25T22:05:05.078967Z","shell.execute_reply.started":"2023-06-25T22:05:00.162733Z","shell.execute_reply":"2023-06-25T22:05:05.078144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:05.080144Z","iopub.execute_input":"2023-06-25T22:05:05.080692Z","iopub.status.idle":"2023-06-25T22:05:05.182309Z","shell.execute_reply.started":"2023-06-25T22:05:05.080658Z","shell.execute_reply":"2023-06-25T22:05:05.181300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nnull1 = dataset_df_1.isnull().sum().sort_values(ascending=False)/len(dataset_df_1)\nnull2 = dataset_df_2.isnull().sum().sort_values(ascending=False)/len(dataset_df_2)\nnull3 = dataset_df_3.isnull().sum().sort_values(ascending=False)/len(dataset_df_3)\n\ndrop1 = list(null1[null1 > 0.9].index)\ndrop2 = list(null2[null2 > 0.9].index)\ndrop3 = list(null3[null3 > 0.9].index)\n\nprint(len(drop1), len(drop2), len(drop3))\n\nfor col in tqdm(dataset_df_1.columns):\n    if dataset_df_1[col].nunique() == 1:\n        #print(col)\n        drop1.append(col)\nfor col in tqdm(dataset_df_2.columns):\n    if dataset_df_2[col].nunique() == 1:\n        #print(col)\n        drop2.append(col)\nfor col in tqdm(dataset_df_3.columns):\n    if dataset_df_3[col].nunique() == 1:\n        #print(col)\n        drop3.append(col)\n\n\nFEATURES1 = [c for c in dataset_df_1.columns if c not in drop1+['level_group']]\nFEATURES2 = [c for c in dataset_df_2.columns if c not in drop2+['level_group']]\nFEATURES3 = [c for c in dataset_df_3.columns if c not in drop3+['level_group']]\n\nprint('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3), 'features')","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:05.183715Z","iopub.execute_input":"2023-06-25T22:05:05.184022Z","iopub.status.idle":"2023-06-25T22:05:05.460224Z","shell.execute_reply.started":"2023-06-25T22:05:05.183992Z","shell.execute_reply":"2023-06-25T22:05:05.459420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - test_ratio))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ndef getSplits(dataset_df):\n    train_x, test_x = split_dataset(dataset_df,0.1)\n    train_x, valid_x = split_dataset(train_x,0.2)\n    return train_x,valid_x,test_x","metadata":{"id":"OZfTcCJfn2PC","execution":{"iopub.status.busy":"2023-06-25T22:05:05.461472Z","iopub.execute_input":"2023-06-25T22:05:05.462084Z","iopub.status.idle":"2023-06-25T22:05:05.468351Z","shell.execute_reply.started":"2023-06-25T22:05:05.462048Z","shell.execute_reply":"2023-06-25T22:05:05.467353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"id":"UdibIrM-XP5-"}},{"cell_type":"code","source":"# Create an empty dictionary to store the models created for each question.\nmodels_xgb = {}\nmodels_cat = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}\nevaluation_f1_dict ={}","metadata":{"id":"7Brds67Wn2PD","execution":{"iopub.status.busy":"2023-06-25T22:05:05.469662Z","iopub.execute_input":"2023-06-25T22:05:05.470721Z","iopub.status.idle":"2023-06-25T22:05:05.481416Z","shell.execute_reply.started":"2023-06-25T22:05:05.470679Z","shell.execute_reply":"2023-06-25T22:05:05.480442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nacc=[]\nfrom xgboost import plot_importance\nbest_threshold=0.63","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:05.482728Z","iopub.execute_input":"2023-06-25T22:05:05.483741Z","iopub.status.idle":"2023-06-25T22:05:05.957543Z","shell.execute_reply.started":"2023-06-25T22:05:05.483706Z","shell.execute_reply":"2023-06-25T22:05:05.956361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport lightgbm as lgbm\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nxgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 6000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:05.959019Z","iopub.execute_input":"2023-06-25T22:05:05.959423Z","iopub.status.idle":"2023-06-25T22:05:07.994464Z","shell.execute_reply.started":"2023-06-25T22:05:05.959385Z","shell.execute_reply":"2023-06-25T22:05:07.993438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier, Pool\ncat_params = {\n        'iterations': 1000,\n        'early_stopping_rounds': 90,\n        'depth': 5,\n        'learning_rate': 0.02,\n        'loss_function': \"Logloss\",\n        'random_seed': 222222,\n        'metric_period': 1,\n        'subsample': 0.8,\n        'colsample_bylevel': 0.4,\n        'verbose': 0,\n        'l2_leaf_reg': 20,\n    }","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:05:07.995953Z","iopub.execute_input":"2023-06-25T22:05:07.996305Z","iopub.status.idle":"2023-06-25T22:05:08.248207Z","shell.execute_reply.started":"2023-06-25T22:05:07.996263Z","shell.execute_reply":"2023-06-25T22:05:08.247247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nmodels_xgb = {}\nvalids_idx = {}\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    \n    if grp == '0-4':\n        df = dataset_df_1\n        FEATURES = FEATURES1\n    if grp == '5-12':\n        df = dataset_df_2\n        FEATURES = FEATURES2\n    if grp == '13-22':\n        df = dataset_df_3\n        FEATURES = FEATURES3\n    print(\"### q_no\", q_no, \"grp\", grp, \"feats : \",len(FEATURES))\n        \n    split = list(GroupKFold(5).split(df.index.unique(), groups = df.index.unique()))\n    \n    y_preds = []\n    for fold, (train_idx, valid_idx) in enumerate(split):\n        \n        # Filter the rows in the datasets based on the selected level group. \n        train_df = df.iloc[train_idx]\n        train_users = train_df.index.values\n        valid_df = df.iloc[valid_idx]\n        valid_users = valid_df.index.values\n        \n        valids_idx[f'{grp}_{q_no}_{fold}'] = valid_idx\n\n\n        # Select the labels for the related q_no.\n        train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n        valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n        X_train = train_df.loc[:, train_df.columns != 'level_group']\n        y_train = train_labels[\"correct\"]\n\n        X_val = valid_df.loc[:, valid_df.columns != 'level_group']\n        y_val = valid_labels[\"correct\"]\n\n        xgbm = XGBClassifier(**xgb_params)\n        #catm = CatBoostClassifier(**cat_params)\n\n        xgbm.fit(X_train[FEATURES].astype('float32'), y_train,\n                    eval_set=[ (X_val[FEATURES].astype('float32'), y_val) ],verbose=0)\n        #catm.fit(X_train[FEATURES].astype('float32'), y_train,\n         #           eval_set=[ (X_val[FEATURES].astype('float32'), y_val) ],verbose=0)\n\n        # Store the model\n        models_xgb[f'{grp}_{q_no}_{fold}'] = xgbm\n        print(\"Done for \",grp,q_no,fold)\n\n        \n    #prediction_df.loc[test_users, q_no-1] = y_pred_val   ","metadata":{"id":"VBO3VCOJn2PF","execution":{"iopub.status.busy":"2023-06-25T22:05:08.249698Z","iopub.execute_input":"2023-06-25T22:05:08.250270Z","iopub.status.idle":"2023-06-25T22:07:13.128040Z","shell.execute_reply.started":"2023-06-25T22:05:08.250221Z","shell.execute_reply":"2023-06-25T22:07:13.127146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc=[]\npr=[]\nrec=[]\nf1=[]\nALL_USERS = dataset_df_1.index.values\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n    if grp == '0-4':\n        df = dataset_df_1\n        FEATURES = FEATURES1\n    if grp == '5-12':\n        df = dataset_df_2\n        FEATURES = FEATURES2\n    if grp == '13-22':\n        df = dataset_df_3\n        FEATURES = FEATURES3\n        \n    #split = list(GroupKFold(5).split(df.index.unique(), groups = df.index.unique()))\n    \n    \n    y_preds = []\n    \n    for fold in range(5):\n        \n        valid_idx = valids_idx[f'{grp}_{q_no}_{fold}']\n        \n        valid_df = df.iloc[valid_idx]\n        valid_users = valid_df.index.values\n        y_val = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n        \n        xgbm = models_xgb[f'{grp}_{q_no}_{fold}']\n\n        y_pred_val_xgb=xgbm.predict_proba(valid_df[FEATURES])\n        y_pred_val=y_pred_val_xgb[:,1]\n        y_preds.append(y_pred_val)\n        oof.loc[valid_users, q_no-1] = y_pred_val\n        \n    #y_pred_val1 = np.mean(y_preds,axis=0)\n    #oof.loc[valid_users, t-1] = y_pred_val1\n    #y_pred_val1=(y_pred_val1 > best_threshold).astype(int).flatten()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:07:13.129170Z","iopub.execute_input":"2023-06-25T22:07:13.129683Z","iopub.status.idle":"2023-06-25T22:07:14.648015Z","shell.execute_reply.started":"2023-06-25T22:07:13.129649Z","shell.execute_reply":"2023-06-25T22:07:14.647185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:07:14.649269Z","iopub.execute_input":"2023-06-25T22:07:14.649763Z","iopub.status.idle":"2023-06-25T22:07:14.676609Z","shell.execute_reply.started":"2023-06-25T22:07:14.649730Z","shell.execute_reply":"2023-06-25T22:07:14.675524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nacc=[]\n# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values\n    \n##################\n# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:07:14.681520Z","iopub.execute_input":"2023-06-25T22:07:14.681831Z","iopub.status.idle":"2023-06-25T22:07:14.888814Z","shell.execute_reply.started":"2023-06-25T22:07:14.681801Z","shell.execute_reply":"2023-06-25T22:07:14.887585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Best threshold \", best_threshold, \"\\tF1 score \", best_score)","metadata":{"id":"qPOfPkm7n2PG","execution":{"iopub.status.busy":"2023-06-25T22:07:14.890284Z","iopub.execute_input":"2023-06-25T22:07:14.890589Z","iopub.status.idle":"2023-06-25T22:07:14.895728Z","shell.execute_reply.started":"2023-06-25T22:07:14.890557Z","shell.execute_reply":"2023-06-25T22:07:14.894759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##0.6971997907477581","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:07:14.897108Z","iopub.execute_input":"2023-06-25T22:07:14.897424Z","iopub.status.idle":"2023-06-25T22:07:14.906993Z","shell.execute_reply.started":"2023-06-25T22:07:14.897394Z","shell.execute_reply":"2023-06-25T22:07:14.906021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nprint('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:07:14.908354Z","iopub.execute_input":"2023-06-25T22:07:14.908656Z","iopub.status.idle":"2023-06-25T22:07:14.942917Z","shell.execute_reply.started":"2023-06-25T22:07:14.908626Z","shell.execute_reply":"2023-06-25T22:07:14.941886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Overall F1 = 0.6936789566795002","metadata":{"execution":{"iopub.status.busy":"2023-06-25T22:07:14.944331Z","iopub.execute_input":"2023-06-25T22:07:14.944647Z","iopub.status.idle":"2023-06-25T22:07:14.948470Z","shell.execute_reply.started":"2023-06-25T22:07:14.944616Z","shell.execute_reply":"2023-06-25T22:07:14.947485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nHere you'll use the `best_threshold` calculate in the previous cell","metadata":{"id":"ezA40GQ4n2PH"}},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\ndfs = {}\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    grp = test.level_group.values[0]\n    session_id = test.session_id.values[0]\n    ##test=dataText(test)\n    \n    feats_sel = feats_sel_1\n    \n    if grp == '0-4':\n        FEATURES = FEATURES1\n        feats_sel = feats_sel_1\n    if grp == '5-12':\n        FEATURES = FEATURES2\n        feats_sel = feats_sel_2\n    if grp == '13-22':\n        FEATURES = FEATURES3\n        feats_sel = feats_sel_3\n\n    sess_ids = test[\"session_id\"].unique()\n    df = pd.DataFrame()\n\n    for sess_id in sess_ids:\n        df_sess = test[test['session_id']==sess_id]\n        if grp == \"0-4\":\n            dfs[sess_id] = df_sess\n        else:\n            if sess_id in dfs:\n                dfs[sess_id] = pd.concat([dfs[sess_id],df_sess])\n            else:\n                dfs[sess_id] = df_sess\n        df=df.append(dfs[sess_id])\n        #print(len(df))\n\n    gc.collect()\n    df = df.sort_values(['session_id','index'])\n\n    test_df = feature_engineer(df,feats_sel)\n    \n    \n    a,b = limits[grp]\n    for t in range(a,b):\n    \n        test_ds = test_df.loc[:, test_df.columns != 'level_group']\n        \n        preds = []\n        for fold in range(5):\n            xgbm = models_xgb[f'{grp}_{t}_{fold}']\n            predictions_xgb = xgbm.predict_proba(test_ds[FEATURES])\n            predictions_xgb=predictions_xgb[:,1]\n            preds.append(predictions_xgb)\n            \n        predictions = np.mean(preds,axis=0)\n        \n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)\n    \n    if grp == '13-22':\n        for sess_id in sess_ids:\n            if sess_id in dfs:\n                del dfs[sess_id]\n                \n    del test,test_df,df\n    gc.collect()","metadata":{"id":"gHiXTnTVn2PI","execution":{"iopub.status.busy":"2023-06-25T22:07:14.949835Z","iopub.execute_input":"2023-06-25T22:07:14.950126Z","iopub.status.idle":"2023-06-25T22:07:24.919478Z","shell.execute_reply.started":"2023-06-25T22:07:14.950097Z","shell.execute_reply":"2023-06-25T22:07:24.918430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"id":"iYBXokAyn2PI","execution":{"iopub.status.busy":"2023-06-25T22:07:24.920628Z","iopub.execute_input":"2023-06-25T22:07:24.923499Z","iopub.status.idle":"2023-06-25T22:07:25.935245Z","shell.execute_reply.started":"2023-06-25T22:07:24.923460Z","shell.execute_reply":"2023-06-25T22:07:25.933895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}