{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"},{"sourceId":5675701,"sourceType":"datasetVersion","datasetId":3244175}],"dockerImageVersionId":30458,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Predict Student Performance from Game Play**\n\n### Ho Chi Minh City University of Science\n\n#### 21KHDL - Intelligent Data Analysis\n\n#### Lecturers:\n- Mr. Nguyễn Tiến Huy\n- Mr. Nguyễn Trần Duy Minh\n- Mr. Lê Thanh Tùng\n\n#### Student:\n- 21127038 - Võ Phú Hãn","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"text-align: center;\"> Student Performance from Game Play - Model Building</h1>","metadata":{}},{"cell_type":"markdown","source":"# 0. Main idea\n\nI'll train GroupKFold models with XGBoost baseline for each question in the game. The models will use all previously data to predict the correctness of the session for the current question, including events that occurred in the corresponding levels. Additionally, new features will be created to improve the model.","metadata":{}},{"cell_type":"markdown","source":"# 1. Import the Required Libraries","metadata":{"id":"zAXHC6-Tn2O5"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport gc\nimport polars as pl\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nfrom xgboost import plot_importance\nimport lightgbm as lgbm\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"id":"IanlX-Eqn2O5","execution":{"iopub.status.busy":"2024-12-06T15:37:14.789924Z","iopub.execute_input":"2024-12-06T15:37:14.790860Z","iopub.status.idle":"2024-12-06T15:37:17.423303Z","shell.execute_reply.started":"2024-12-06T15:37:14.790822Z","shell.execute_reply":"2024-12-06T15:37:17.422118Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Preprocessing Data for Training","metadata":{}},{"cell_type":"markdown","source":"## Load and Prepare Train Data\n\nLoad `train.csv` with defined scheme.","metadata":{}},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'str',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'\n    }\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes) ###, nrows=500000)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:33:37.204312Z","iopub.execute_input":"2024-12-06T10:33:37.204732Z","iopub.status.idle":"2024-12-06T10:35:15.813380Z","shell.execute_reply.started":"2024-12-06T10:33:37.204661Z","shell.execute_reply":"2024-12-06T10:35:15.812126Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:35:15.815244Z","iopub.execute_input":"2024-12-06T10:35:15.815586Z","iopub.status.idle":"2024-12-06T10:35:15.992910Z","shell.execute_reply.started":"2024-12-06T10:35:15.815541Z","shell.execute_reply":"2024-12-06T10:35:15.991832Z"},"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Load `train_labels.cv`","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"id":"KD4uayl2n2O9","execution":{"iopub.status.busy":"2024-12-06T10:35:15.993902Z","iopub.execute_input":"2024-12-06T10:35:15.994205Z","iopub.status.idle":"2024-12-06T10:35:16.347720Z","shell.execute_reply.started":"2024-12-06T10:35:15.994178Z","shell.execute_reply":"2024-12-06T10:35:16.346749Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Split `session_id` into `session` and `q` features","metadata":{}},{"cell_type":"code","source":"labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"id":"Kva8_Dbqn2O9","execution":{"iopub.status.busy":"2024-12-06T10:35:16.350116Z","iopub.execute_input":"2024-12-06T10:35:16.350439Z","iopub.status.idle":"2024-12-06T10:35:17.060767Z","shell.execute_reply.started":"2024-12-06T10:35:16.350410Z","shell.execute_reply":"2024-12-06T10:35:17.059663Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the first 5 examples\nlabels.head(5)","metadata":{"id":"0eD-KZMvn2O-","execution":{"iopub.status.busy":"2024-12-06T10:35:17.062236Z","iopub.execute_input":"2024-12-06T10:35:17.062565Z","iopub.status.idle":"2024-12-06T10:35:17.075788Z","shell.execute_reply.started":"2024-12-06T10:35:17.062530Z","shell.execute_reply":"2024-12-06T10:35:17.074573Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are 3 level groups. Every questions of each group will happen after a level checkpoint (4, 12, 22). The main idea is split data needed for training into 3 groups for each checkpoint:\n- `level_group` == '0-4'   $\\Rightarrow$ `question` 1->3\n- `level_group` == '5-12'  $\\Rightarrow$ `question` 4->13\n- `level_group` == '13-22' $\\Rightarrow$ `question` 14->18\n\nAnd because, we're going to use all previous data in a session to predict the correction, therefore, I'll split the train data to 3 dataframe:\n- `dataset_df_1` within `level` 0->4\n- `dataset_df_2` within `level` 0->12\n- `dataset_df_3` within `level` 0->22","metadata":{}},{"cell_type":"code","source":"dataset_df_1 = dataset_df[dataset_df.level_group == '0-4']\ndataset_df_2 = dataset_df[dataset_df.level_group != '13-22']\ndataset_df_3 = dataset_df\nprint(f\"dataset_df_1: {dataset_df_1.shape}\")\nprint(f\"dataset_df_2: {dataset_df_2.shape}\")\nprint(f\"dataset_df_3: {dataset_df_3.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:35:17.077411Z","iopub.execute_input":"2024-12-06T10:35:17.077781Z","iopub.status.idle":"2024-12-06T10:35:18.714535Z","shell.execute_reply.started":"2024-12-06T10:35:17.077749Z","shell.execute_reply":"2024-12-06T10:35:18.713434Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del dataset_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:35:18.716009Z","iopub.execute_input":"2024-12-06T10:35:18.716336Z","iopub.status.idle":"2024-12-06T10:35:18.865630Z","shell.execute_reply.started":"2024-12-06T10:35:18.716305Z","shell.execute_reply":"2024-12-06T10:35:18.864579Z"},"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df_1.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:35:18.866751Z","iopub.execute_input":"2024-12-06T10:35:18.867135Z","iopub.status.idle":"2024-12-06T10:35:18.879933Z","shell.execute_reply.started":"2024-12-06T10:35:18.867104Z","shell.execute_reply":"2024-12-06T10:35:18.878778Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Prepare Feature Selection Data","metadata":{}},{"cell_type":"code","source":"feats=pd.read_csv(\"/kaggle/input/featur/feature_sort.csv\")\nfeats.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:35:18.881527Z","iopub.execute_input":"2024-12-06T10:35:18.881918Z","iopub.status.idle":"2024-12-06T10:35:19.038269Z","shell.execute_reply.started":"2024-12-06T10:35:18.881862Z","shell.execute_reply":"2024-12-06T10:35:19.037160Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This dataset, sourced from [Vadim Kamaev's notebook](https://www.kaggle.com/code/vadimkamaev/catboost-new/notebook), is generated to support analysis and model training by capturing detailed relationships between features for this particular competition. For a deeper understanding of how this dataset was constructed, refer to [this link](https://www.kaggle.com/code/vadimkamaev/feature-sort).\n\nI'll explain some *key features* used in this model building:\n- `quest`: questions (from 1 to 18).  \n- (`col1`, `val1`), (`col2`, `val2`), (`col3`, `val3`): pairs of column names and their corresponding unique values (`col1` and `col2` are categorical columns; `col3` is a numerical column).  \n- `kol_col`: the number of columns involved in generating the feature (1 or 2).  \n- `kach`: the quality score for each feature, likely based on F1-score, calculated during training or validation.\n\nI'll select features with `kach>10` and split into 3 datasets for 3 train datasets above.","metadata":{}},{"cell_type":"code","source":"\n# Select features with kach > 10\nfeats_sel=feats[feats['kach']>10]\nfeats_sel_1=feats_sel[feats_sel['quest']<=3]\nfeats_sel_2=feats_sel[feats_sel['quest']<=12]\nfeats_sel_3=feats_sel\n\n# Get unique feature-value pairs\ncols=['kol_col','col1','val1','col2','val2']\nfeats_sel_1=feats_sel_1[cols].drop_duplicates().reset_index(drop=True)\nfeats_sel_2=feats_sel_2[cols].drop_duplicates().reset_index(drop=True)\nfeats_sel_3=feats_sel_3[cols].drop_duplicates().reset_index(drop=True)\n\nprint(len(feats_sel_1))\nprint(len(feats_sel_2))\nprint(len(feats_sel_3))\nfeats_sel_1.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:35:19.041886Z","iopub.execute_input":"2024-12-06T10:35:19.042221Z","iopub.status.idle":"2024-12-06T10:35:19.072598Z","shell.execute_reply.started":"2024-12-06T10:35:19.042190Z","shell.execute_reply":"2024-12-06T10:35:19.071784Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Feature Engineering\n\nGenerate new features to prepare for model building","metadata":{}},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['level','page','room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y','delt_time_next']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:35:19.073514Z","iopub.execute_input":"2024-12-06T10:35:19.073890Z","iopub.status.idle":"2024-12-06T10:35:19.078832Z","shell.execute_reply.started":"2024-12-06T10:35:19.073856Z","shell.execute_reply":"2024-12-06T10:35:19.077927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineer(train, feats_sel):\n    \"\"\"\n    This function performs feature engineering on a given training dataset by creating new features \n    based on session aggregations and conditional groupings.\n\n    Parameters:\n    train (DataFrame): The input dataset containing session data.\n    feats_sel (DataFrame): A configuration DataFrame specifying conditions for additional features.\n\n    Returns:\n    DataFrame: A new DataFrame with engineered features.\n    \"\"\"\n\n    # Sort by session and elapsed time to ensure chronological order\n    train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n\n    # Calculate time differences between consecutive rows\n    train['d_time'] = train['elapsed_time'].diff(1)\n    train['d_time'].fillna(0, inplace=True)  # Replace NaN values in the first row\n    train['delt_time'] = train['d_time'].clip(0, 103000)  # Clip large time differences\n    train['delt_time_next'] = train['delt_time'].shift(-1)  # Shift for the next time delta\n\n    # Create a new DataFrame to store session features\n    new_train = pd.DataFrame(index=train['session_id'].unique())  \n    new_train['session_id'] = new_train.index  \n\n    # # Re-sort to ensure consistent ordering (this appears redundant but may be intentional for safety)\n    # train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n\n    # # Recalculate time differences\n    # train['d_time'] = train['elapsed_time'].diff(1)\n    # train['d_time'].fillna(0, inplace=True)\n    # train['delt_time'] = train['d_time'].clip(0, 103000)\n    # train['delt_time_next'] = train['delt_time'].shift(-1)\n\n    # Base feature generation (categorical and numerical aggregates)\n    base = True  # Toggle for base feature generation\n    if base:\n        for c in CATEGORICAL:\n            # Count unique values per session for categorical columns\n            new_train[f'{c}_nunique'] = train.groupby(['session_id'])[c].agg('nunique')\n        for c in NUMERICAL:\n            # Compute mean and sum per session for numerical columns\n            new_train[f'{c}_mean'] = train.groupby(['session_id'])[c].agg('mean')\n            new_train[f'{c}_sum'] = train.groupby(['session_id'])[c].agg('sum')\n\n    # Session counts and time quantiles\n    new_train['session_index_count'] = train.groupby(['session_id'])['index'].count()\n    new_train['d_time_q3'] = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    new_train['d_time_q8'] = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    new_train['d_time_q5'] = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    new_train['d_time_q65'] = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n\n    # Hover duration statistics\n    new_train['hover_duration_mean'] = train.groupby(['session_id'])['hover_duration'].agg('mean')\n    new_train['hover_duration_std'] = train.groupby(['session_id'])['hover_duration'].agg('std') \n\n    # Delta time statistics\n    new_train['delt_time_mean'] = train.groupby(['session_id'])['delt_time'].agg('mean')\n    new_train['delt_time_std'] = train.groupby(['session_id'])['delt_time'].agg('std') \n    new_train['delt_time_max'] = train.groupby(['session_id'])['delt_time'].agg('max')\n    new_train['delt_time_min'] = train.groupby(['session_id'])['delt_time'].agg('min') \n\n    # Extracting date and time components from session_id\n    new_train['year'] = new_train['session_id'].apply(lambda x: int(str(x)[:2])).astype(np.uint8)  # Extract year\n    new_train['month'] = new_train['session_id'].apply(lambda x: int(str(x)[2:4]) + 1).astype(np.uint8)  # Extract month\n    new_train['day'] = new_train['session_id'].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)  # Extract day\n    new_train['sess_time'] = new_train['session_id'].apply(lambda x: int(str(x)[6:8])).astype(np.uint8) + \\\n                             new_train['session_id'].apply(lambda x: int(str(x)[8:10])).astype(np.uint8) / 60  # Hour and minute\n\n    # Fill NaN values with -1 for consistency\n    new_train = new_train.fillna(-1)\n\n    # Conditional feature generation (based on feats_sel configuration)\n    t1 = feats_sel[feats_sel['kol_col'] == 1]  # Single-condition features\n    for i in range(len(t1)):\n        col1 = t1['col1'].iloc[i]\n        val1 = t1['val1'].iloc[i]\n\n        # Create mask based on the condition\n        maska1 = (train[col1] == val1)\n        \n        # Generate features for sessions matching the condition\n        new_train[f'{col1}_{hash(val1)}_delt_time_next_sum'] = train[maska1].groupby(['session_id'])['delt_time_next'].sum()\n        new_train[f'{col1}_{hash(val1)}_delt_time_mean'] = train[maska1].groupby(['session_id'])['delt_time'].mean()\n        new_train[f'{col1}_{hash(val1)}_index_count'] = train[maska1].groupby(['session_id'])['index'].count()\n\n    t2 = feats_sel[feats_sel['kol_col'] == 2]  # Two-condition features\n    for i in range(len(t2)):\n        col1 = t2['col1'].iloc[i]\n        val1 = t2['val1'].iloc[i]\n        col2 = t2['col2'].iloc[i]\n        val2 = t2['val2'].iloc[i]\n\n        # Create mask based on the combined conditions\n        maska2 = (train[col1] == val1) & (train[col2] == val2)\n        \n        # Generate features for sessions matching the conditions\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_next_sum'] = train[maska2].groupby(['session_id'])['delt_time_next'].sum()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_delt_time_mean'] = train[maska2].groupby(['session_id'])['delt_time'].mean()\n        new_train[f'{col1}_{hash(val1)}_{col2}_{hash(val2)}_index_count'] = train[maska2].groupby(['session_id'])['index'].count()\n\n    return new_train\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:35:19.079933Z","iopub.execute_input":"2024-12-06T10:35:19.080247Z","iopub.status.idle":"2024-12-06T10:35:19.108741Z","shell.execute_reply.started":"2024-12-06T10:35:19.080216Z","shell.execute_reply":"2024-12-06T10:35:19.107478Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Apply function `feature_engineer()` for datasets","metadata":{}},{"cell_type":"code","source":"dataset_df_1 = feature_engineer(dataset_df_1,feats_sel_1)\nprint(\"Full prepared dataset_df_1 shape: {}\".format(dataset_df_1.shape))\ndataset_df_1.head()","metadata":{"id":"JKcoPoemn2PA","execution":{"iopub.status.busy":"2024-12-06T10:35:19.110107Z","iopub.execute_input":"2024-12-06T10:35:19.110439Z","iopub.status.idle":"2024-12-06T10:35:34.390760Z","shell.execute_reply.started":"2024-12-06T10:35:19.110409Z","shell.execute_reply":"2024-12-06T10:35:34.389834Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:35:34.391721Z","iopub.execute_input":"2024-12-06T10:35:34.392017Z","iopub.status.idle":"2024-12-06T10:35:34.547647Z","shell.execute_reply.started":"2024-12-06T10:35:34.391988Z","shell.execute_reply":"2024-12-06T10:35:34.546304Z"},"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df_2 = feature_engineer(dataset_df_2,feats_sel_2)\nprint(\"Full prepared dataset_df_2 shape is {}\".format(dataset_df_2.shape))\ndataset_df_2.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:35:34.549049Z","iopub.execute_input":"2024-12-06T10:35:34.549424Z","iopub.status.idle":"2024-12-06T10:37:10.107994Z","shell.execute_reply.started":"2024-12-06T10:35:34.549381Z","shell.execute_reply":"2024-12-06T10:37:10.106673Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:37:10.109754Z","iopub.execute_input":"2024-12-06T10:37:10.110111Z","iopub.status.idle":"2024-12-06T10:37:10.260015Z","shell.execute_reply.started":"2024-12-06T10:37:10.110079Z","shell.execute_reply":"2024-12-06T10:37:10.258836Z"},"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_df_3 = feature_engineer(dataset_df_3,feats_sel_3)\nprint(\"Full prepared dataset_df_3 shape is {}\".format(dataset_df_3.shape))\ndataset_df_3.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:37:10.261022Z","iopub.execute_input":"2024-12-06T10:37:10.261376Z","iopub.status.idle":"2024-12-06T10:41:01.011086Z","shell.execute_reply.started":"2024-12-06T10:37:10.261344Z","shell.execute_reply":"2024-12-06T10:41:01.010065Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:41:01.012201Z","iopub.execute_input":"2024-12-06T10:41:01.012501Z","iopub.status.idle":"2024-12-06T10:41:01.181191Z","shell.execute_reply.started":"2024-12-06T10:41:01.012474Z","shell.execute_reply":"2024-12-06T10:41:01.180082Z"},"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Drop columns with high missing ratio or only have 1 value.","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Calculate the percentage of missing values for each column in the datasets\nnull1 = dataset_df_1.isnull().sum().sort_values(ascending=False) / len(dataset_df_1)\nnull2 = dataset_df_2.isnull().sum().sort_values(ascending=False) / len(dataset_df_2)\nnull3 = dataset_df_3.isnull().sum().sort_values(ascending=False) / len(dataset_df_3)\n\n# Identify columns with more than 90% missing values to drop\ndrop1 = list(null1[null1 > 0.9].index)\ndrop2 = list(null2[null2 > 0.9].index)\ndrop3 = list(null3[null3 > 0.9].index)\n\n# # Print the number of columns to be dropped due to high missing values\n# print(len(drop1), len(drop2), len(drop3))\n\n# Loop through each column in the first dataset to identify columns with a single unique value\nfor col in tqdm(dataset_df_1.columns):\n    if dataset_df_1[col].nunique() == 1:\n        # Add columns with only one unique value to the drop list\n        drop1.append(col)\n\n# Repeat the process for the second dataset\nfor col in tqdm(dataset_df_2.columns):\n    if dataset_df_2[col].nunique() == 1:\n        drop2.append(col)\n\n# Repeat the process for the third dataset\nfor col in tqdm(dataset_df_3.columns):\n    if dataset_df_3[col].nunique() == 1:\n        drop3.append(col)\n\n# Create a list of features for training, excluding the dropped columns and 'level_group'\nFEATURES1 = [c for c in dataset_df_1.columns if c not in drop1 + ['level_group']]\nFEATURES2 = [c for c in dataset_df_2.columns if c not in drop2 + ['level_group']]\nFEATURES3 = [c for c in dataset_df_3.columns if c not in drop3 + ['level_group']]\n\n# Print the number of features that will be used for training for each dataset\n# print('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3), 'features')\nprint(f\"We'll train 'dataset_df_1' with {len(FEATURES1)} features\")\nprint(f\"We'll train 'dataset_df_2' with {len(FEATURES2)} features\")\nprint(f\"We'll train 'dataset_df_3' with {len(FEATURES3)} features\")","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:44:07.766422Z","iopub.execute_input":"2024-12-06T10:44:07.767150Z","iopub.status.idle":"2024-12-06T10:44:08.640394Z","shell.execute_reply.started":"2024-12-06T10:44:07.767096Z","shell.execute_reply":"2024-12-06T10:44:08.639710Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Model Building & Training\n\nTrain one XGB model for each question with 5-fold cross-validation","metadata":{}},{"cell_type":"markdown","source":"### XGBoost Introduction\n**XGBoost** is a highly efficient and scalable machine learning algorithm based on gradient boosting. It was developed by Tianqi Chen and has become one of the most popular algorithms for structured/tabular data. XGBoost is known for its performance and speed, making it a go-to tool for data science competitions (like Kaggle).\n\n### How it work\n**XGBoost** works by building an ensemble of decision trees sequentially, where each tree corrects the errors of the previous one. It uses gradient boosting, optimizing the model by minimizing a loss function through gradient descent. During training, XGBoost adjusts the weights of misclassified samples and adds regularization to avoid overfitting. The trees are built iteratively, with each new tree focusing on the residuals (errors) of the previous trees, ultimately improving the model’s predictions.\n\n### Pros\n- **High Performance:** XGBoost is known for its speed and accuracy, often winning Kaggle competitions.\n- **Handles Missing Data:** It can handle missing values in the dataset naturally.\n- **Parallelization:** Supports parallel and distributed computing, making it faster than other gradient boosting algorithms.\n- **Regularization:** Built-in regularization (L1 and L2) helps prevent overfitting.\n- **Flexibility:** Works for both regression and classification problems.\n- **Scalability:** Can handle large datasets efficiently due to optimizations.\n\n### Cons\n- **Complexity:** Tuning the model can be complex with many hyperparameters.\n- **Interpretability:** Like most ensemble methods, XGBoost models are less interpretable compared to simpler models (e.g., linear regression).\n- **Memory Consumption:** For very large datasets, it can consume a significant amount of memory.\n- **Overfitting Risk:** While regularization helps, overfitting can still occur if hyperparameters are not properly tuned.\n","metadata":{}},{"cell_type":"markdown","source":"Set XGBoost parameters","metadata":{}},{"cell_type":"code","source":"xgb_params = {\n    'objective': 'binary:logistic',  # Specifies binary classification (logistic regression)\n    'eval_metric': 'logloss',        # Evaluation metric used during training (log loss for binary classification)\n    'learning_rate': 0.05,           # Step size during training (smaller values require more boosting rounds)\n    'max_depth': 4,                  # Maximum depth of trees (helps control model complexity)\n    'n_estimators': 6000,            # Number of boosting rounds (trees) to build\n    'early_stopping_rounds': 50,     # Stop training early if validation score doesn't improve for 50 rounds\n    'tree_method': 'hist',           # Tree-building method (histogram-based method for faster training)\n    'subsample': 0.8,                # Fraction of samples to use for each tree (helps prevent overfitting)\n    'colsample_bytree': 0.4,         # Fraction of features to use for each tree (helps with regularization)\n    'use_label_encoder': False       # Disables label encoding for the target variable (to avoid warnings)\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:44:12.631883Z","iopub.execute_input":"2024-12-06T10:44:12.632298Z","iopub.status.idle":"2024-12-06T10:44:12.643170Z","shell.execute_reply.started":"2024-12-06T10:44:12.632263Z","shell.execute_reply":"2024-12-06T10:44:12.641661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from catboost import CatBoostClassifier, Pool\n# cat_params = {\n#         'iterations': 1000,\n#         'early_stopping_rounds': 90,\n#         'depth': 5,\n#         'learning_rate': 0.02,\n#         'loss_function': \"Logloss\",\n#         'random_seed': 222222,\n#         'metric_period': 1,\n#         'subsample': 0.8,\n#         'colsample_bylevel': 0.4,\n#         'verbose': 0,\n#         'l2_leaf_reg': 20,\n#     }","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:44:12.647903Z","iopub.execute_input":"2024-12-06T10:44:12.648497Z","iopub.status.idle":"2024-12-06T10:44:12.655368Z","shell.execute_reply.started":"2024-12-06T10:44:12.648457Z","shell.execute_reply":"2024-12-06T10:44:12.653952Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Traing XGB models with 5-fold cross-validation for 18 questions","metadata":{}},{"cell_type":"code","source":"models_xgb = {}\nvalids_idx = {}\n\n# Iterate through questions 1 to 18 to train models for each question\nfor q_no in range(1,19):\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n\n    # Select the appropriate dataset and features based on the group (grp)\n    if grp == '0-4':\n        df = dataset_df_1\n        FEATURES = FEATURES1\n    if grp == '5-12':\n        df = dataset_df_2\n        FEATURES = FEATURES2\n    if grp == '13-22':\n        df = dataset_df_3\n        FEATURES = FEATURES3\n    print(\"### q_no\", q_no, \"grp\", grp, \"feats : \",len(FEATURES))\n\n    # Generate indices for the training and validation sets for each fold\n    split = list(GroupKFold(5).split(df.index.unique(), groups = df.index.unique()))\n    \n    # Perform 5-fold cross-validation using GroupKFold\n    for fold, (train_idx, valid_idx) in enumerate(split):\n        \n        # Filter the rows in the datasets based on train and valid indices\n        train_df = df.iloc[train_idx]\n        train_users = train_df.index.values\n        valid_df = df.iloc[valid_idx]\n        valid_users = valid_df.index.values\n\n        # Store the validation indices for later use\n        valids_idx[f'{grp}_{q_no}_{fold}'] = valid_idx\n\n\n        # Select the labels for the related q_no and session.\n        train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n        valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n        # Prepare the features (X) and target (y) for training and validation\n        X_train = train_df.loc[:, train_df.columns != 'level_group']\n        y_train = train_labels[\"correct\"]\n        X_val = valid_df.loc[:, valid_df.columns != 'level_group']\n        y_val = valid_labels[\"correct\"]\n        \n        # Train model\n        xgbm = XGBClassifier(**xgb_params)\n        #catm = CatBoostClassifier(**cat_params)\n\n        xgbm.fit(X_train[FEATURES].astype('float32'), y_train,\n                    eval_set=[ (X_val[FEATURES].astype('float32'), y_val) ],verbose=0)\n        # catm.fit(X_train[FEATURES].astype('float32'), y_train,\n        #          eval_set=[ (X_val[FEATURES].astype('float32'), y_val) ],verbose=0)\n\n        # Store the model\n        models_xgb[f'{grp}_{q_no}_{fold}'] = xgbm\n        print(\"Done for \",grp,q_no,fold)","metadata":{"id":"VBO3VCOJn2PF","execution":{"iopub.status.busy":"2024-12-06T10:44:12.665652Z","iopub.execute_input":"2024-12-06T10:44:12.666108Z","iopub.status.idle":"2024-12-06T10:56:03.104443Z","shell.execute_reply.started":"2024-12-06T10:44:12.666071Z","shell.execute_reply":"2024-12-06T10:56:03.103569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Out-of-Fold Prediction","metadata":{}},{"cell_type":"code","source":"ALL_USERS = dataset_df_1.index.values # get all users/sessions\n\n# Initialize an empty DataFrame to store out-of-fold (OOF) predictions for all sessions and questions\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\n\n# Loop through each question number (1 to 18)\nfor q_no in range(1, 19):\n    # Select level group for the question based on the q_no.\n    if q_no <= 3: \n        grp = '0-4' \n    elif q_no <= 13: \n        grp = '5-12'\n    elif q_no <= 22: \n        grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n\n    # Select the appropriate dataset and features based on the group (grp)\n    if grp == '0-4':\n        df = dataset_df_1 \n        FEATURES = FEATURES1 \n    if grp == '5-12':\n        df = dataset_df_2  \n        FEATURES = FEATURES2\n    if grp == '13-22':\n        df = dataset_df_3\n        FEATURES = FEATURES3\n\n    y_preds = [] # strore validation predicts\n\n    # Loop through 5-fold cross-validation\n    for fold in range(5):\n        \n        # Extract the validation dataframe using stored validation indices\n        valid_idx = valids_idx[f'{grp}_{q_no}_{fold}']\n        valid_df = df.iloc[valid_idx]\n        valid_users = valid_df.index.values\n        \n        # # Get the true labels for the validation set\n        # y_val = labels.loc[labels.q == q_no].set_index('session').loc[valid_users]\n        \n        # Load the pre-trained XGBoost model\n        xgbm = models_xgb[f'{grp}_{q_no}_{fold}']\n\n        # Predict probabilities for the validation set\n        y_pred_val_xgb = xgbm.predict_proba(valid_df[FEATURES])\n        y_pred_val = y_pred_val_xgb[:, 1]\n        \n        y_preds.append(y_pred_val)\n        \n        # Store the predictions in the OOF DataFrame for the valid sessions and current question number\n        oof.loc[valid_users, q_no - 1] = y_pred_val\n\n        # # Optionally, you could compute the average of the predictions across all folds\n        # y_pred_val1 = np.mean(y_preds, axis=0)\n        # oof.loc[valid_users, t - 1] = y_pred_val1\n        \n        # If using a threshold, you could classify the predictions as 0 or 1 based on the threshold\n        # y_pred_val1 = (y_pred_val1 > best_threshold).astype(int).flatten()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T10:56:03.105906Z","iopub.execute_input":"2024-12-06T10:56:03.106194Z","iopub.status.idle":"2024-12-06T10:56:07.785724Z","shell.execute_reply.started":"2024-12-06T10:56:03.106166Z","shell.execute_reply":"2024-12-06T10:56:07.784785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:56:12.442791Z","iopub.execute_input":"2024-12-06T10:56:12.443070Z","iopub.status.idle":"2024-12-06T10:56:12.473556Z","shell.execute_reply.started":"2024-12-06T10:56:12.443044Z","shell.execute_reply":"2024-12-06T10:56:12.472463Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Threshold Optimization\nFind the best threshold performance","metadata":{}},{"cell_type":"code","source":"true = oof.copy()\n\n# Populate the `true` DataFrame with the actual labels (ground truth) for all sessions\nfor k in range(18):\n    # Extract true labels for each question and align them with the sessions\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values\n\n##################\n# Optimize the threshold for converting probabilities into binary predictions (0 or 1)\n\nscores = []\nthresholds = []\n\nbest_score = 0\nbest_threshold = 0\n\nprint(\"List of threshoolds: \")\n# Iterate over thresholds\nfor threshold in np.arange(0.4, 0.81, 0.01):\n    print(f'{threshold:.02f}, ', end='')\n\n    # Apply the threshold to the OOF probabilities to generate binary predictions\n    preds = (oof.values.reshape((-1)) > threshold).astype('int')\n\n    # Calculate the F1 score for the binary predictions using the true labels\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')\n\n    scores.append(m)\n    thresholds.append(threshold)\n\n    # Update the best score and threshold\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:56:12.474836Z","iopub.execute_input":"2024-12-06T10:56:12.475129Z","iopub.status.idle":"2024-12-06T10:56:19.862976Z","shell.execute_reply.started":"2024-12-06T10:56:12.475100Z","shell.execute_reply":"2024-12-06T10:56:19.861542Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Best threshold \", best_threshold, \"\\tF1 score \", best_score)","metadata":{"id":"qPOfPkm7n2PG","execution":{"iopub.status.busy":"2024-12-06T10:56:19.864325Z","iopub.execute_input":"2024-12-06T10:56:19.864638Z","iopub.status.idle":"2024-12-06T10:56:19.872595Z","shell.execute_reply.started":"2024-12-06T10:56:19.864609Z","shell.execute_reply":"2024-12-06T10:56:19.871544Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Evaluation","metadata":{}},{"cell_type":"code","source":"print('When using optimal threshold...')\n\n# Evaluate F1 scores for each individual question\nfor k in range(18):\n    # Apply the optimal threshold to the OOF predictions for question `k`\n    # Convert probabilities to binary predictions (0 or 1) using the `best_threshold`\n    # Compute the macro F1 score between the true labels and binary predictions\n    m = f1_score(true[k].values, (oof[k].values > best_threshold).astype('int'), average='macro')\n    \n    print(f'Q{k}: F1 =', m)\n\n# Evaluate the overall F1 score across all questions\n# Flatten the true labels and predictions into 1D arrays for overall evaluation\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1)) > best_threshold).astype('int'), average='macro')\n\nprint('==> Overall F1 =', m)","metadata":{"execution":{"iopub.status.busy":"2024-12-06T10:56:19.887462Z","iopub.execute_input":"2024-12-06T10:56:19.887872Z","iopub.status.idle":"2024-12-06T10:56:20.243475Z","shell.execute_reply.started":"2024-12-06T10:56:19.887841Z","shell.execute_reply":"2024-12-06T10:56:20.242132Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Submission\n\nHere I'll use the `best_threshold` calculate in the previous cell","metadata":{"id":"ezA40GQ4n2PH"}},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\ndfs = {}\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    grp = test.level_group.values[0]\n    session_id = test.session_id.values[0]\n    ##test=dataText(test)\n    \n    feats_sel = feats_sel_1\n    \n    if grp == '0-4':\n        FEATURES = FEATURES1\n        feats_sel = feats_sel_1\n    if grp == '5-12':\n        FEATURES = FEATURES2\n        feats_sel = feats_sel_2\n    if grp == '13-22':\n        FEATURES = FEATURES3\n        feats_sel = feats_sel_3\n\n    sess_ids = test[\"session_id\"].unique()\n    df = pd.DataFrame()\n\n    for sess_id in sess_ids:\n        df_sess = test[test['session_id']==sess_id]\n        if grp == \"0-4\":\n            dfs[sess_id] = df_sess\n        else:\n            if sess_id in dfs:\n                dfs[sess_id] = pd.concat([dfs[sess_id],df_sess])\n            else:\n                dfs[sess_id] = df_sess\n        df=df.append(dfs[sess_id])\n        #print(len(df))\n\n    gc.collect()\n    df = df.sort_values(['session_id','index'])\n\n    test_df = feature_engineer(df,feats_sel)\n    \n    \n    a,b = limits[grp]\n    for t in range(a,b):\n    \n        test_ds = test_df.loc[:, test_df.columns != 'level_group']\n        \n        preds = []\n        for fold in range(5):\n            xgbm = models_xgb[f'{grp}_{t}_{fold}']\n            predictions_xgb = xgbm.predict_proba(test_ds[FEATURES])\n            predictions_xgb=predictions_xgb[:,1]\n            preds.append(predictions_xgb)\n            \n        predictions = np.mean(preds,axis=0)\n        \n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)\n    \n    if grp == '13-22':\n        for sess_id in sess_ids:\n            if sess_id in dfs:\n                del dfs[sess_id]\n                \n    del test,test_df,df\n    gc.collect()","metadata":{"id":"gHiXTnTVn2PI","execution":{"iopub.status.busy":"2024-12-06T10:56:20.253093Z","iopub.execute_input":"2024-12-06T10:56:20.253502Z","iopub.status.idle":"2024-12-06T10:56:30.284983Z","shell.execute_reply.started":"2024-12-06T10:56:20.253457Z","shell.execute_reply":"2024-12-06T10:56:30.283930Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! head submission.csv","metadata":{"id":"iYBXokAyn2PI","execution":{"iopub.status.busy":"2024-12-06T10:56:30.286084Z","iopub.execute_input":"2024-12-06T10:56:30.286354Z","iopub.status.idle":"2024-12-06T10:56:31.581009Z","shell.execute_reply.started":"2024-12-06T10:56:30.286327Z","shell.execute_reply":"2024-12-06T10:56:31.579749Z"},"trusted":true},"outputs":[],"execution_count":null}]}