{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import the Required Libraries","metadata":{"id":"zAXHC6-Tn2O5"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport gc\nimport polars as pl","metadata":{"id":"IanlX-Eqn2O5","execution":{"iopub.status.busy":"2023-06-26T06:45:30.702968Z","iopub.execute_input":"2023-06-26T06:45:30.704812Z","iopub.status.idle":"2023-06-26T06:45:30.712530Z","shell.execute_reply.started":"2023-06-26T06:45:30.704716Z","shell.execute_reply":"2023-06-26T06:45:30.711174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.715017Z","iopub.execute_input":"2023-06-26T06:45:30.715916Z","iopub.status.idle":"2023-06-26T06:45:30.727555Z","shell.execute_reply.started":"2023-06-26T06:45:30.715704Z","shell.execute_reply":"2023-06-26T06:45:30.726542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats=pd.read_csv(\"/kaggle/input/featur/feature_sort.csv\")\nfeats_sel=feats[feats['kach']>10]\nfeats_sel_1=feats_sel[feats_sel['quest']<=3]\nfeats_sel_2=feats_sel[feats_sel['quest']<=12]\nfeats_sel_3=feats_sel\n\ncols=['kol_col','col1','val1','col2','val2']\nfeats_sel_1=feats_sel_1[cols].drop_duplicates().reset_index(drop=True)\nfeats_sel_2=feats_sel_2[cols].drop_duplicates().reset_index(drop=True)\nfeats_sel_3=feats_sel_3[cols].drop_duplicates().reset_index(drop=True)\n\nprint(len(feats_sel_1))\nprint(len(feats_sel_2))\nprint(len(feats_sel_3))\nfeats_sel_1.head()","metadata":{"id":"_XItl24kn2O7","execution":{"iopub.status.busy":"2023-06-26T06:45:30.729214Z","iopub.execute_input":"2023-06-26T06:45:30.729794Z","iopub.status.idle":"2023-06-26T06:45:30.852258Z","shell.execute_reply.started":"2023-06-26T06:45:30.729756Z","shell.execute_reply":"2023-06-26T06:45:30.851462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_sel_3","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.854385Z","iopub.execute_input":"2023-06-26T06:45:30.854896Z","iopub.status.idle":"2023-06-26T06:45:30.875826Z","shell.execute_reply.started":"2023-06-26T06:45:30.854863Z","shell.execute_reply":"2023-06-26T06:45:30.873627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gold_feats=pd.read_csv(\"/kaggle/input/student-perf-get-feats/golden_feats.csv\")\ngold_feats.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.877428Z","iopub.execute_input":"2023-06-26T06:45:30.878360Z","iopub.status.idle":"2023-06-26T06:45:30.909149Z","shell.execute_reply.started":"2023-06-26T06:45:30.878301Z","shell.execute_reply":"2023-06-26T06:45:30.906205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"---Before---\")\nprint(len(feats_sel_1))\nprint(len(feats_sel_2))\nprint(len(feats_sel_3))\n\nfeats_sel_1=gold_feats[gold_feats.grp=='0-4'][['kol_col','col1','val1','col2','val2']].drop_duplicates().reset_index(drop=True)\nfeats_sel_2=gold_feats[gold_feats.grp!='13-22'][['kol_col','col1','val1','col2','val2']].drop_duplicates().reset_index(drop=True)\nfeats_sel_3=gold_feats[['kol_col','col1','val1','col2','val2']].drop_duplicates().reset_index(drop=True)\n\nprint(\"---After---\")\nprint(len(feats_sel_1))\nprint(len(feats_sel_2))\nprint(len(feats_sel_3))","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.910704Z","iopub.execute_input":"2023-06-26T06:45:30.911268Z","iopub.status.idle":"2023-06-26T06:45:30.928930Z","shell.execute_reply.started":"2023-06-26T06:45:30.911228Z","shell.execute_reply":"2023-06-26T06:45:30.927962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['level','page','room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y','delt_time_next']","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.931462Z","iopub.execute_input":"2023-06-26T06:45:30.931934Z","iopub.status.idle":"2023-06-26T06:45:30.938873Z","shell.execute_reply.started":"2023-06-26T06:45:30.931891Z","shell.execute_reply":"2023-06-26T06:45:30.936811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train,feats_sel):\n    \n    train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    train['d_time'] = train['elapsed_time'].diff(1)\n    train['d_time'].fillna(0, inplace=True)\n    train['delt_time'] = train['d_time'].clip(0, 103000)\n    train['delt_time_next'] = train['delt_time'].shift(-1)\n\n    new_train = pd.DataFrame(index=train['session_id'].unique())  \n    new_train['session_id'] = new_train.index \n    \n    train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    \n\n    train['d_time'] = train['elapsed_time'].diff(1)\n    train['d_time'].fillna(0, inplace=True)\n    train['delt_time'] = train['d_time'].clip(0, 103000)\n    train['delt_time_next'] = train['delt_time'].shift(-1)\n    \n###\n    base=True\n    if base:\n        for c in CATEGORICAL:\n            new_train[f'{c}_nunique'] = train.groupby(['session_id'])[c].agg('nunique')\n        for c in NUMERICAL:\n            new_train[f'{c}_mean'] = train.groupby(['session_id'])[c].agg('mean')\n            new_train[f'{c}_sum'] = train.groupby(['session_id'])[c].agg('sum')\n            \n            \n###\n    new_train['session_index_count'] = train.groupby(['session_id'])['index'].count()\n    new_train['d_time_q3'] = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    new_train['d_time_q8'] = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    new_train['d_time_q5'] = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    new_train['d_time_q65'] = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    \n    new_train['hover_duration_mean'] = train.groupby(['session_id'])['hover_duration'].agg('mean')\n    new_train['hover_duration_std'] = train.groupby(['session_id'])['hover_duration'].agg('std') \n    new_train['delt_time_mean'] = train.groupby(['session_id'])['delt_time'].agg('mean')\n    new_train['delt_time_std'] = train.groupby(['session_id'])['delt_time'].agg('std') \n    new_train['delt_time_max'] = train.groupby(['session_id'])['delt_time'].agg('max')\n    new_train['delt_time_min'] = train.groupby(['session_id'])['delt_time'].agg('min') \n    \n    new_train['year'] = new_train['session_id'].apply(lambda x: int(str(x)[:2])).astype(np.uint8) # \"year\"\n    new_train['month'] = new_train['session_id'].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8) # \"month\"\n    new_train['day'] = new_train['session_id'].apply(lambda x: int(str(x)[4:6])).astype(np.uint8) # \"day\"\n    new_train['sess_time'] = new_train['session_id'].apply(lambda x: int(str(x)[6:8])).astype(np.uint8) + new_train['session_id'].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)/60\n    new_train = new_train.fillna(-1)\n    \n    t1=feats_sel[feats_sel['kol_col']==1]\n    for i in range(len(t1)):\n        col1 = t1['col1'].iloc[i]\n        val1 = t1['val1'].iloc[i]\n\n        maska1 = (train[col1] == val1)\n        new_train[f'{col1}_{hash(val1)}_delt_time_next_sum'] = train[maska1].groupby(['session_id'])['delt_time_next'].sum()\n        new_train[f'{col1}_{hash(val1)}_delt_time_mean'] = train[maska1].groupby(['session_id'])['delt_time'].mean()\n        new_train[f'{col1}_{hash(val1)}_delt_time_std'] = train[maska1].groupby(['session_id'])['delt_time'].std()\n        new_train[f'{col1}_{hash(val1)}_index_count'] = train[maska1].groupby(['session_id'])['index'].count()\n\n    t2=feats_sel[feats_sel['kol_col']==2]\n    for i in range(len(t2)):\n        col1 = t2['col1'].iloc[i]\n        val1 = t2['val1'].iloc[i]\n        col2 = t2['col2'].iloc[i]\n        val2 = t2['val2'].iloc[i]\n\n        maska2 = (train[col1] == val1) & (train[col2] == val2)\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_next_sum'] = train[maska2].groupby(['session_id'])['delt_time_next'].sum()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_mean'] = train[maska2].groupby(['session_id'])['delt_time'].mean()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_std'] = train[maska2].groupby(['session_id'])['delt_time'].std()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_index_count'] = train[maska2].groupby(['session_id'])['index'].count()\n    \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.940891Z","iopub.execute_input":"2023-06-26T06:45:30.942789Z","iopub.status.idle":"2023-06-26T06:45:30.971501Z","shell.execute_reply.started":"2023-06-26T06:45:30.942675Z","shell.execute_reply":"2023-06-26T06:45:30.970262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'str',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'\n    }\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes) ##, nrows=500000)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:30.975446Z","iopub.execute_input":"2023-06-26T06:45:30.975823Z","iopub.status.idle":"2023-06-26T06:45:31.939717Z","shell.execute_reply.started":"2023-06-26T06:45:30.975790Z","shell.execute_reply":"2023-06-26T06:45:31.938611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:31.941180Z","iopub.execute_input":"2023-06-26T06:45:31.941430Z","iopub.status.idle":"2023-06-26T06:45:32.435184Z","shell.execute_reply.started":"2023-06-26T06:45:31.941405Z","shell.execute_reply":"2023-06-26T06:45:32.433767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"id":"KD4uayl2n2O9","execution":{"iopub.status.busy":"2023-06-26T06:45:32.436849Z","iopub.execute_input":"2023-06-26T06:45:32.437318Z","iopub.status.idle":"2023-06-26T06:45:32.696404Z","shell.execute_reply.started":"2023-06-26T06:45:32.437282Z","shell.execute_reply":"2023-06-26T06:45:32.695303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"id":"Kva8_Dbqn2O9","execution":{"iopub.status.busy":"2023-06-26T06:45:32.697958Z","iopub.execute_input":"2023-06-26T06:45:32.698258Z","iopub.status.idle":"2023-06-26T06:45:33.321898Z","shell.execute_reply.started":"2023-06-26T06:45:32.698232Z","shell.execute_reply":"2023-06-26T06:45:33.320439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first 5 examples\nlabels.head(5)","metadata":{"id":"0eD-KZMvn2O-","execution":{"iopub.status.busy":"2023-06-26T06:45:33.323337Z","iopub.execute_input":"2023-06-26T06:45:33.324032Z","iopub.status.idle":"2023-06-26T06:45:33.334576Z","shell.execute_reply.started":"2023-06-26T06:45:33.323997Z","shell.execute_reply":"2023-06-26T06:45:33.333329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_1 = dataset_df[dataset_df.level_group=='0-4']\ndataset_df_2 = dataset_df[dataset_df.level_group!='13-22']\ndataset_df_3 = dataset_df\nprint(dataset_df_1.shape)\nprint(dataset_df_2.shape)\nprint(dataset_df_3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:33.336059Z","iopub.execute_input":"2023-06-26T06:45:33.336579Z","iopub.status.idle":"2023-06-26T06:45:33.373227Z","shell.execute_reply.started":"2023-06-26T06:45:33.336542Z","shell.execute_reply":"2023-06-26T06:45:33.372148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del dataset_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:33.374592Z","iopub.execute_input":"2023-06-26T06:45:33.374883Z","iopub.status.idle":"2023-06-26T06:45:33.957205Z","shell.execute_reply.started":"2023-06-26T06:45:33.374856Z","shell.execute_reply":"2023-06-26T06:45:33.956492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_1 = feature_engineer(dataset_df_1,feats_sel_1)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df_1.shape))\ndataset_df_1.head()","metadata":{"id":"JKcoPoemn2PA","execution":{"iopub.status.busy":"2023-06-26T06:45:33.958323Z","iopub.execute_input":"2023-06-26T06:45:33.958751Z","iopub.status.idle":"2023-06-26T06:45:34.874606Z","shell.execute_reply.started":"2023-06-26T06:45:33.958709Z","shell.execute_reply":"2023-06-26T06:45:34.873281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:34.876256Z","iopub.execute_input":"2023-06-26T06:45:34.876569Z","iopub.status.idle":"2023-06-26T06:45:35.349200Z","shell.execute_reply.started":"2023-06-26T06:45:34.876542Z","shell.execute_reply":"2023-06-26T06:45:35.347567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_2 = feature_engineer(dataset_df_2,feats_sel_2)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df_2.shape))\ndataset_df_2.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:35.350871Z","iopub.execute_input":"2023-06-26T06:45:35.351560Z","iopub.status.idle":"2023-06-26T06:45:38.141864Z","shell.execute_reply.started":"2023-06-26T06:45:35.351517Z","shell.execute_reply":"2023-06-26T06:45:38.140476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:38.143211Z","iopub.execute_input":"2023-06-26T06:45:38.143518Z","iopub.status.idle":"2023-06-26T06:45:38.623408Z","shell.execute_reply.started":"2023-06-26T06:45:38.143486Z","shell.execute_reply":"2023-06-26T06:45:38.622615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df_3 = feature_engineer(dataset_df_3,feats_sel_3)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df_3.shape))\ndataset_df_3.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:38.624671Z","iopub.execute_input":"2023-06-26T06:45:38.625179Z","iopub.status.idle":"2023-06-26T06:45:43.820103Z","shell.execute_reply.started":"2023-06-26T06:45:38.625150Z","shell.execute_reply":"2023-06-26T06:45:43.819260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:43.821371Z","iopub.execute_input":"2023-06-26T06:45:43.821862Z","iopub.status.idle":"2023-06-26T06:45:44.315542Z","shell.execute_reply.started":"2023-06-26T06:45:43.821835Z","shell.execute_reply":"2023-06-26T06:45:44.314075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nnull1 = dataset_df_1.isnull().sum().sort_values(ascending=False)/len(dataset_df_1)\nnull2 = dataset_df_2.isnull().sum().sort_values(ascending=False)/len(dataset_df_2)\nnull3 = dataset_df_3.isnull().sum().sort_values(ascending=False)/len(dataset_df_3)\n\ndrop1 = list(null1[null1 > 0.9].index)\ndrop2 = list(null2[null2 > 0.9].index)\ndrop3 = list(null3[null3 > 0.9].index)\n\nprint(len(drop1), len(drop2), len(drop3))\n\nfor col in tqdm(dataset_df_1.columns):\n    if dataset_df_1[col].nunique() == 1:\n        #print(col)\n        drop1.append(col)\nfor col in tqdm(dataset_df_2.columns):\n    if dataset_df_2[col].nunique() == 1:\n        #print(col)\n        drop2.append(col)\nfor col in tqdm(dataset_df_3.columns):\n    if dataset_df_3[col].nunique() == 1:\n        #print(col)\n        drop3.append(col)\n\n\nFEATURES1 = [c for c in dataset_df_1.columns if c not in drop1+['level_group']]\nFEATURES2 = [c for c in dataset_df_2.columns if c not in drop2+['level_group']]\nFEATURES3 = [c for c in dataset_df_3.columns if c not in drop3+['level_group']]\n\nprint('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3), 'features')","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:44.317225Z","iopub.execute_input":"2023-06-26T06:45:44.317527Z","iopub.status.idle":"2023-06-26T06:45:44.738762Z","shell.execute_reply.started":"2023-06-26T06:45:44.317497Z","shell.execute_reply":"2023-06-26T06:45:44.737380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - test_ratio))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ndef getSplits(dataset_df):\n    train_x, test_x = split_dataset(dataset_df,0.1)\n    train_x, valid_x = split_dataset(train_x,0.2)\n    return train_x,valid_x,test_x","metadata":{"id":"OZfTcCJfn2PC","execution":{"iopub.status.busy":"2023-06-26T06:45:44.740421Z","iopub.execute_input":"2023-06-26T06:45:44.740762Z","iopub.status.idle":"2023-06-26T06:45:44.747272Z","shell.execute_reply.started":"2023-06-26T06:45:44.740720Z","shell.execute_reply":"2023-06-26T06:45:44.746030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"id":"UdibIrM-XP5-"}},{"cell_type":"code","source":"# Create an empty dictionary to store the models created for each question.\nmodels_xgb = {}\nmodels_cat = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}\nevaluation_f1_dict ={}","metadata":{"id":"7Brds67Wn2PD","execution":{"iopub.status.busy":"2023-06-26T06:45:44.752495Z","iopub.execute_input":"2023-06-26T06:45:44.752878Z","iopub.status.idle":"2023-06-26T06:45:44.799123Z","shell.execute_reply.started":"2023-06-26T06:45:44.752844Z","shell.execute_reply":"2023-06-26T06:45:44.797917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nacc=[]\nfrom xgboost import plot_importance\nbest_threshold=0.63","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:44.800485Z","iopub.execute_input":"2023-06-26T06:45:44.801603Z","iopub.status.idle":"2023-06-26T06:45:44.806290Z","shell.execute_reply.started":"2023-06-26T06:45:44.801536Z","shell.execute_reply":"2023-06-26T06:45:44.805428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport lightgbm as lgbm\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nxgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 6000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:44.807512Z","iopub.execute_input":"2023-06-26T06:45:44.808352Z","iopub.status.idle":"2023-06-26T06:45:44.821546Z","shell.execute_reply.started":"2023-06-26T06:45:44.808317Z","shell.execute_reply":"2023-06-26T06:45:44.819901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier, Pool\ncat_params = {\n        'iterations': 1000,\n        'early_stopping_rounds': 90,\n        'depth': 5,\n        'learning_rate': 0.02,\n        'loss_function': \"Logloss\",\n        'random_seed': 222222,\n        'metric_period': 1,\n        'subsample': 0.8,\n        'colsample_bylevel': 0.4,\n        'verbose': 0,\n        'l2_leaf_reg': 20,\n    }","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:45:44.823083Z","iopub.execute_input":"2023-06-26T06:45:44.823477Z","iopub.status.idle":"2023-06-26T06:45:44.840242Z","shell.execute_reply.started":"2023-06-26T06:45:44.823438Z","shell.execute_reply":"2023-06-26T06:45:44.838604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nmodels_xgb = {}\nvalids_idx = {}\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    \n    if grp == '0-4':\n        df = dataset_df_1\n        FEATURES = FEATURES1\n    if grp == '5-12':\n        df = dataset_df_2\n        FEATURES = FEATURES2\n    if grp == '13-22':\n        df = dataset_df_3\n        FEATURES = FEATURES3\n    print(\"### q_no\", q_no, \"grp\", grp, \"feats : \",len(FEATURES))\n        \n    split = list(GroupKFold(5).split(df.index.unique(), groups = df.index.unique()))\n    \n    y_preds = []\n    for fold, (train_idx, valid_idx) in enumerate(split):\n        \n        # Filter the rows in the datasets based on the selected level group. \n        train_df = df.iloc[train_idx]\n        train_users = train_df.index.values\n        valid_df = df.iloc[valid_idx]\n        valid_users = valid_df.index.values\n        \n        valids_idx[f'{grp}_{q_no}_{fold}'] = valid_idx\n\n\n        # Select the labels for the related q_no.\n        train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n        valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n        X_train = train_df.loc[:, train_df.columns != 'level_group']\n        y_train = train_labels[\"correct\"]\n\n        X_val = valid_df.loc[:, valid_df.columns != 'level_group']\n        y_val = valid_labels[\"correct\"]\n\n        xgbm = XGBClassifier(**xgb_params)\n        #catm = CatBoostClassifier(**cat_params)\n\n        xgbm.fit(X_train[FEATURES].astype('float32'), y_train,\n                    eval_set=[ (X_val[FEATURES].astype('float32'), y_val) ],verbose=0)\n        #catm.fit(X_train[FEATURES].astype('float32'), y_train,\n         #           eval_set=[ (X_val[FEATURES].astype('float32'), y_val) ],verbose=0)\n\n        # Store the model\n        models_xgb[f'{grp}_{q_no}_{fold}'] = xgbm\n        print(\"Done for \",grp,q_no,fold)\n\n        \n    #prediction_df.loc[test_users, q_no-1] = y_pred_val   ","metadata":{"id":"VBO3VCOJn2PF","execution":{"iopub.status.busy":"2023-06-26T06:45:44.841755Z","iopub.execute_input":"2023-06-26T06:45:44.842154Z","iopub.status.idle":"2023-06-26T06:47:47.975237Z","shell.execute_reply.started":"2023-06-26T06:45:44.842114Z","shell.execute_reply":"2023-06-26T06:47:47.973701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc=[]\npr=[]\nrec=[]\nf1=[]\nALL_USERS = dataset_df_1.index.values\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n    if grp == '0-4':\n        df = dataset_df_1\n        FEATURES = FEATURES1\n    if grp == '5-12':\n        df = dataset_df_2\n        FEATURES = FEATURES2\n    if grp == '13-22':\n        df = dataset_df_3\n        FEATURES = FEATURES3\n        \n    #split = list(GroupKFold(5).split(df.index.unique(), groups = df.index.unique()))\n    \n    \n    y_preds = []\n    \n    for fold in range(5):\n        \n        valid_idx = valids_idx[f'{grp}_{q_no}_{fold}']\n        \n        valid_df = df.iloc[valid_idx]\n        valid_users = valid_df.index.values\n        y_val = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n        \n        xgbm = models_xgb[f'{grp}_{q_no}_{fold}']\n\n        y_pred_val_xgb=xgbm.predict_proba(valid_df[FEATURES])\n        y_pred_val=y_pred_val_xgb[:,1]\n        y_preds.append(y_pred_val)\n        oof.loc[valid_users, q_no-1] = y_pred_val\n        \n","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:47:47.977150Z","iopub.execute_input":"2023-06-26T06:47:47.977861Z","iopub.status.idle":"2023-06-26T06:47:49.697459Z","shell.execute_reply.started":"2023-06-26T06:47:47.977816Z","shell.execute_reply":"2023-06-26T06:47:49.695985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:47:49.698770Z","iopub.execute_input":"2023-06-26T06:47:49.699065Z","iopub.status.idle":"2023-06-26T06:47:49.729471Z","shell.execute_reply.started":"2023-06-26T06:47:49.699038Z","shell.execute_reply":"2023-06-26T06:47:49.728216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nacc=[]\n# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values\n    \n##################\n# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:47:49.730529Z","iopub.execute_input":"2023-06-26T06:47:49.730859Z","iopub.status.idle":"2023-06-26T06:47:49.902880Z","shell.execute_reply.started":"2023-06-26T06:47:49.730828Z","shell.execute_reply":"2023-06-26T06:47:49.901821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Best threshold \", best_threshold, \"\\tF1 score \", best_score)","metadata":{"id":"qPOfPkm7n2PG","execution":{"iopub.status.busy":"2023-06-26T06:47:49.904053Z","iopub.execute_input":"2023-06-26T06:47:49.904326Z","iopub.status.idle":"2023-06-26T06:47:49.908743Z","shell.execute_reply.started":"2023-06-26T06:47:49.904300Z","shell.execute_reply":"2023-06-26T06:47:49.907929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##0.6971997907477581","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:47:49.909721Z","iopub.execute_input":"2023-06-26T06:47:49.910499Z","iopub.status.idle":"2023-06-26T06:47:49.920681Z","shell.execute_reply.started":"2023-06-26T06:47:49.910436Z","shell.execute_reply":"2023-06-26T06:47:49.919431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nprint('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:47:49.921979Z","iopub.execute_input":"2023-06-26T06:47:49.922652Z","iopub.status.idle":"2023-06-26T06:47:49.955266Z","shell.execute_reply.started":"2023-06-26T06:47:49.922623Z","shell.execute_reply":"2023-06-26T06:47:49.954067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Overall F1 = 0.6936789566795002","metadata":{"execution":{"iopub.status.busy":"2023-06-26T06:47:49.956343Z","iopub.execute_input":"2023-06-26T06:47:49.956863Z","iopub.status.idle":"2023-06-26T06:47:49.961221Z","shell.execute_reply.started":"2023-06-26T06:47:49.956831Z","shell.execute_reply":"2023-06-26T06:47:49.960266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nHere you'll use the `best_threshold` calculate in the previous cell","metadata":{"id":"ezA40GQ4n2PH"}},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\ndfs = {}\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    grp = test.level_group.values[0]\n    session_id = test.session_id.values[0]\n    ##test=dataText(test)\n    \n    feats_sel = feats_sel_1\n    \n    if grp == '0-4':\n        FEATURES = FEATURES1\n        feats_sel = feats_sel_1\n    if grp == '5-12':\n        FEATURES = FEATURES2\n        feats_sel = feats_sel_2\n    if grp == '13-22':\n        FEATURES = FEATURES3\n        feats_sel = feats_sel_3\n\n    sess_ids = test[\"session_id\"].unique()\n    df = pd.DataFrame()\n\n    for sess_id in sess_ids:\n        df_sess = test[test['session_id']==sess_id]\n        if grp == \"0-4\":\n            dfs[sess_id] = df_sess\n        else:\n            if sess_id in dfs:\n                dfs[sess_id] = pd.concat([dfs[sess_id],df_sess])\n            else:\n                dfs[sess_id] = df_sess\n        df=df.append(dfs[sess_id])\n        #print(len(df))\n\n    gc.collect()\n    df = df.sort_values(['session_id','index'])\n\n    test_df = feature_engineer(df,feats_sel)\n    \n    \n    a,b = limits[grp]\n    for t in range(a,b):\n    \n        test_ds = test_df.loc[:, test_df.columns != 'level_group']\n        \n        preds = []\n        for fold in range(5):\n            xgbm = models_xgb[f'{grp}_{t}_{fold}']\n            predictions_xgb = xgbm.predict_proba(test_ds[FEATURES])\n            predictions_xgb=predictions_xgb[:,1]\n            preds.append(predictions_xgb)\n            \n        predictions = np.mean(preds,axis=0)\n        \n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)\n    \n    if grp == '13-22':\n        for sess_id in sess_ids:\n            if sess_id in dfs:\n                del dfs[sess_id]\n                \n    del test,test_df,df\n    gc.collect()","metadata":{"id":"gHiXTnTVn2PI","execution":{"iopub.status.busy":"2023-06-26T06:47:49.963174Z","iopub.execute_input":"2023-06-26T06:47:49.963947Z","iopub.status.idle":"2023-06-26T06:47:49.998881Z","shell.execute_reply.started":"2023-06-26T06:47:49.963902Z","shell.execute_reply":"2023-06-26T06:47:49.997959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"id":"iYBXokAyn2PI","execution":{"iopub.status.busy":"2023-06-26T06:47:49.999602Z","iopub.status.idle":"2023-06-26T06:47:50.000651Z","shell.execute_reply.started":"2023-06-26T06:47:50.000281Z","shell.execute_reply":"2023-06-26T06:47:50.000326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}