{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30806,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Constrain\nNote: this competition is aimed at producing models that are small and lightweight. We have introduced compute constraints to match - your VMs will have only 2 CPUs, 8GB of RAM, and no GPU available. You will still have a maximum of 9 hours to complete the task, but between the constraints and the efficiency prize there will be some interesting sub-problems to solve. Good luck!","metadata":{}},{"cell_type":"markdown","source":"# Import require libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2025-12-22T03:41:33.984069Z","iopub.execute_input":"2025-12-22T03:41:33.984450Z","iopub.status.idle":"2025-12-22T03:41:34.397725Z","shell.execute_reply.started":"2025-12-22T03:41:33.984418Z","shell.execute_reply":"2025-12-22T03:41:34.396433Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the Dataset\n#### With their correct data types","metadata":{}},{"cell_type":"code","source":"# Define the dtypes for columns to optimize memory usage\ndtypes = {\n    'elapsed_time': np.int32,\n    'event_name': 'category',\n    'name': 'category',\n    'level': np.uint8,\n    'room_coor_x': np.float32,\n    'room_coor_y': np.float32,\n    'screen_coor_x': np.float32,\n    'screen_coor_y': np.float32,\n    'hover_duration': np.float32,\n    'text': 'category',\n    'fqid': 'category',\n    'room_fqid': 'category',\n    'text_fqid': 'category',\n    'fullscreen': 'category',\n    'hq': 'category',\n    'music': 'category',\n    'level_group': 'category'\n}\n\n# Initialize an empty list to store chunks\nchunks = []\n\n# Load the data in chunks\nchunk_size = 500000  # Define the chunk size (adjust as needed)\nfor chunk in pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes, chunksize=chunk_size):\n    chunks.append(chunk)\n\n# Concatenate all chunks into a single DataFrame\ndataset_df = pd.concat(chunks, axis=0)\n\n# Print the shape of the full dataset\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2025-12-22T03:41:34.400795Z","iopub.execute_input":"2025-12-22T03:41:34.401321Z","iopub.status.idle":"2025-12-22T03:43:57.805921Z","shell.execute_reply.started":"2025-12-22T03:41:34.401283Z","shell.execute_reply":"2025-12-22T03:43:57.804888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the first 5 examples\ndataset_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:43:57.807175Z","iopub.execute_input":"2025-12-22T03:43:57.807495Z","iopub.status.idle":"2025-12-22T03:43:57.837316Z","shell.execute_reply.started":"2025-12-22T03:43:57.807466Z","shell.execute_reply":"2025-12-22T03:43:57.836626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the labels","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:43:57.838932Z","iopub.execute_input":"2025-12-22T03:43:57.839243Z","iopub.status.idle":"2025-12-22T03:43:58.272770Z","shell.execute_reply.started":"2025-12-22T03:43:57.839214Z","shell.execute_reply":"2025-12-22T03:43:58.271872Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Each value in the column, `session_id` is a combination of both the `session` and the `question number`. We will split these into individual columns for ease of use.","metadata":{}},{"cell_type":"code","source":"labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:43:58.274327Z","iopub.execute_input":"2025-12-22T03:43:58.274602Z","iopub.status.idle":"2025-12-22T03:43:58.947972Z","shell.execute_reply.started":"2025-12-22T03:43:58.274576Z","shell.execute_reply":"2025-12-22T03:43:58.947141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the first 5 examples\nlabels.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:43:58.950043Z","iopub.execute_input":"2025-12-22T03:43:58.950351Z","iopub.status.idle":"2025-12-22T03:43:58.959088Z","shell.execute_reply.started":"2025-12-22T03:43:58.950321Z","shell.execute_reply":"2025-12-22T03:43:58.958501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data discovery\n## I. Dataset","metadata":{}},{"cell_type":"markdown","source":"#### Columns\n* session_id - the ID of the session the event took place in\n* index - the index of the event for the session\n* elapsed_time - how much time has passed (in milliseconds) between the start of the session and when the event was recorded\n* event_name - the name of the event type\n* name - the event name (e.g. identifies whether a notebook_click is is opening or closing the notebook)\n* level - what level of the game the event occurred in (0 to 22)\n* page - the page number of the event (only for notebook-related events)\n* room_coor_x - the coordinates of the click in reference to the in-game room (only for click events)\n* room_coor_y - the coordinates of the click in reference to the in-game room (only for click events)\n* screen_coor_x - the coordinates of the click in reference to the player’s screen (only for click events)\n* screen_coor_y - the coordinates of the click in reference to the player’s screen (only for click events)\n* hover_duration - how long (in milliseconds) the hover happened for (only for hover events)\n* text - the text the player sees during this event\n* fqid - the fully qualified ID of the event\n* room_fqid - the fully qualified ID of the room the event took place in\n* text_fqid - the fully qualified ID of the\n* fullscreen - whether the player is in fullscreen mode\n* hq - whether the game is in high-quality\n* music - whether the game music is on or off\n* level_group - which group of levels - and group of questions - this row belongs to (0-4, 5-12, 13-22)\n","metadata":{}},{"cell_type":"markdown","source":"### 1. Dataset at first glance","metadata":{}},{"cell_type":"code","source":"dataset_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:43:58.960026Z","iopub.execute_input":"2025-12-22T03:43:58.960375Z","iopub.status.idle":"2025-12-22T03:44:09.123575Z","shell.execute_reply.started":"2025-12-22T03:43:58.960344Z","shell.execute_reply":"2025-12-22T03:44:09.122461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in dataset_df.columns:\n    print(col + \": \" + str(dataset_df[col].nunique()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:44:09.124601Z","iopub.execute_input":"2025-12-22T03:44:09.124873Z","iopub.status.idle":"2025-12-22T03:44:21.689864Z","shell.execute_reply.started":"2025-12-22T03:44:09.124846Z","shell.execute_reply":"2025-12-22T03:44:21.688166Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Check the proportion of missing data","metadata":{}},{"cell_type":"code","source":"missing_rate = (dataset_df.isna().sum() / len(dataset_df))\nmissing_rate  * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:46:50.399283Z","iopub.execute_input":"2025-12-22T03:46:50.399624Z","iopub.status.idle":"2025-12-22T03:46:53.899427Z","shell.execute_reply.started":"2025-12-22T03:46:50.399595Z","shell.execute_reply":"2025-12-22T03:46:53.898439Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Comment:\n* The columns `page`, `hover_duration`, `text`, and `text_fqid` have a high rate of missing data (above 50%), so we drop them\n* Although column `fqid` has a missing rate at about 30%, i choose to drop it\n* The `level_group` column is not need because we already has the `level` column ","metadata":{}},{"cell_type":"markdown","source":"#### Comment:\n* The columns `room_coor_x`, `room_coor_y`, `screen_coor_x`, and `screen_coor_y` is importance so i will fill it","metadata":{}},{"cell_type":"code","source":"# Create subplots (2 rows, 2 columns)\nfig, ax = plt.subplots(2, 2, figsize=(12, 8))\n\n# Flatten the axes array for easy iteration\naxes = ax.ravel()\n\n# Plot each column in its respective subplot\nfor i, col in enumerate(['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y']):\n    axes[i].hist(dataset_df[col], bins=10, edgecolor='black', alpha=0.7)\n    axes[i].set_title(f'Histogram of {col}')\n    axes[i].set_xlabel('Value')\n    axes[i].set_ylabel('Frequency')\n\n# Adjust layout\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:44:25.197708Z","iopub.execute_input":"2025-12-22T03:44:25.197994Z","iopub.status.idle":"2025-12-22T03:44:27.808063Z","shell.execute_reply.started":"2025-12-22T03:44:25.197965Z","shell.execute_reply":"2025-12-22T03:44:27.807064Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Comment:\n* For `room_coor_x` and `room_coor_y`, use mean since distribution is symmetric.\n* For other skewed features, use median to reduce bias from outliers.","metadata":{}},{"cell_type":"markdown","source":"## II. Labels","metadata":{}},{"cell_type":"markdown","source":"### 1. Labels at first glance","metadata":{}},{"cell_type":"code","source":"labels.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:44:27.809363Z","iopub.execute_input":"2025-12-22T03:44:27.809653Z","iopub.status.idle":"2025-12-22T03:44:27.862040Z","shell.execute_reply.started":"2025-12-22T03:44:27.809625Z","shell.execute_reply":"2025-12-22T03:44:27.861178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. The distribution of each question","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"\\\"Correct\\\" column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    #print(n, str(n))\n    ax = plt.subplot(6, 3, n)\n\n    # filter df and plot ticker on the new subplot axis\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['b', 'c'])\n    \n    # chart formatting\n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:44:27.863324Z","iopub.execute_input":"2025-12-22T03:44:27.863645Z","iopub.status.idle":"2025-12-22T03:44:30.447319Z","shell.execute_reply.started":"2025-12-22T03:44:27.863570Z","shell.execute_reply":"2025-12-22T03:44:30.446172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Comment:\n* The distribution of labels for each question appears to be skewed, with a higher proportion of 'correct' values.","metadata":{}},{"cell_type":"markdown","source":"# Process data","metadata":{}},{"cell_type":"code","source":"NUMERICAL = list(dataset_df.select_dtypes(include=['int32', 'float32']).columns)\nNUMERICAL.pop()\nNUMERICAL","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:48:52.911825Z","iopub.execute_input":"2025-12-22T03:48:52.912472Z","iopub.status.idle":"2025-12-22T03:48:53.163066Z","shell.execute_reply.started":"2025-12-22T03:48:52.912432Z","shell.execute_reply":"2025-12-22T03:48:53.162083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CATEGORICAL = list(dataset_df.select_dtypes(include='object').columns)\nCATEGORICAL","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:44:30.693844Z","iopub.execute_input":"2025-12-22T03:44:30.694167Z","iopub.status.idle":"2025-12-22T03:44:33.967019Z","shell.execute_reply.started":"2025-12-22T03:44:30.694137Z","shell.execute_reply":"2025-12-22T03:44:33.966001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def numerical_feature_engineer(dataset_df):\n    dfs = []\n    for n in NUMERICAL:\n        print(n)\n        tmp = dataset_df.groupby(['session','level_group'], observed=True)[n].agg('mean')\n        dfs.append(tmp)\n\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    # dataset_df = dataset_df.set_index('session')\n    return dataset_df\n\ndef feature_engineer(dataset_df):\n    \n    new_dataset_df = dataset_df\n    new_dataset_df.rename(columns={'session_id': 'session'}, inplace=True)\n\n    # Let drop all cols with missing rate above 50%\n    cols_missing = missing_rate[missing_rate > 0.3].index\n    new_dataset_df = new_dataset_df.drop(columns=cols_missing)\n    \n    for col in ['room_coor_x', 'room_coor_y']:\n        mean_value = new_dataset_df[col].mean()\n        new_dataset_df[col] = new_dataset_df[col].fillna(value=mean_value)\n\n    for col in ['screen_coor_x', 'screen_coor_y']:\n        median_value = new_dataset_df[col].median()\n        new_dataset_df[col] = new_dataset_df[col].fillna(value=median_value)\n\n    new_dataset_df = numerical_feature_engineer(new_dataset_df)\n    new_dataset_df.set_index('session', inplace=True)\n    return new_dataset_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:44:33.971021Z","iopub.execute_input":"2025-12-22T03:44:33.971385Z","iopub.status.idle":"2025-12-22T03:44:33.981281Z","shell.execute_reply.started":"2025-12-22T03:44:33.971344Z","shell.execute_reply":"2025-12-22T03:44:33.979939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let train models\n* We will use SVM, because I think it suit our task requirement \n* Also, We will have 18 models for 18 question","metadata":{}},{"cell_type":"code","source":"new_dataset_df = feature_engineer(dataset_df)\nnew_dataset_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:48:56.200344Z","iopub.execute_input":"2025-12-22T03:48:56.201254Z","iopub.status.idle":"2025-12-22T03:49:07.239347Z","shell.execute_reply.started":"2025-12-22T03:48:56.201212Z","shell.execute_reply":"2025-12-22T03:49:07.238172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_x = new_dataset_df.iloc[:int(len(new_dataset_df) * 0.8)]\nvalid_x = new_dataset_df.iloc[int(len(new_dataset_df) * 0.8):]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:49:12.977792Z","iopub.execute_input":"2025-12-22T03:49:12.978151Z","iopub.status.idle":"2025-12-22T03:49:12.983038Z","shell.execute_reply.started":"2025-12-22T03:49:12.978120Z","shell.execute_reply":"2025-12-22T03:49:12.982293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.metrics import accuracy_score\n\nmodels = {}\nevaluation_dict = {}\n\nfor q_no in range(1, 19):\n    if q_no <= 3: grp = '0-4'\n    elif q_no <= 13: grp = '5-12'\n    elif q_no <= 22: grp = '13-22'\n    print(f'Processing q_no {q_no} with XGBoost')\n\n    # Data selection\n    train_df = train_x.loc[train_x.level_group == grp].copy()\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp].copy()\n    valid_users = valid_df.index.values\n\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Separate features and target\n    X_train = train_df.drop(columns=[\"level_group\"]) \n    y_train = train_labels[\"correct\"]\n    X_test = valid_df.drop(columns=[\"level_group\"]) \n    y_test = valid_labels[\"correct\"]\n\n    # Initialize XGBoost Classifier\n    # 'scale_pos_weight' is the XGBoost equivalent to SVC's 'balanced' class weight\n    model = xgb.XGBClassifier(\n        n_estimators=100,\n        max_depth=4,\n        learning_rate=0.05,\n        objective='binary:logistic',\n        tree_method='hist',  # Fast histogram optimized training\n        random_state=42\n    )\n\n    # Fit the model\n    model.fit(X_train, y_train)\n\n    # Store and evaluate\n    models[f'{grp}_{q_no}'] = model\n    y_pred = model.predict(X_test)\n    evaluation_dict[q_no] = accuracy_score(y_test, y_pred)\n    \n    print(f\"Question {q_no} Accuracy: {evaluation_dict[q_no]:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:49:12.984430Z","iopub.execute_input":"2025-12-22T03:49:12.984843Z","iopub.status.idle":"2025-12-22T03:49:17.121362Z","shell.execute_reply.started":"2025-12-22T03:49:12.984808Z","shell.execute_reply":"2025-12-22T03:49:17.119494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluation_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:49:17.123989Z","iopub.execute_input":"2025-12-22T03:49:17.124460Z","iopub.status.idle":"2025-12-22T03:49:17.137389Z","shell.execute_reply.started":"2025-12-22T03:49:17.124420Z","shell.execute_reply":"2025-12-22T03:49:17.136087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a, b = limits[grp]\n    \n    for t in range(a, b):\n        model = models[f'{grp}_{t}']\n    \n        # Prepare the test data (same preprocessing as for training)\n        X_test = test_df.drop(columns=[\"level_group\"])  # Drop non-feature columns\n    \n        # Predict using the trained SVM model\n        predictions = model.predict(X_test)\n    \n        # Map predictions to 'correct' values\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask, 'correct'] = predictions\n    # Submit the results\n    env.predict(sample_submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:49:28.029654Z","iopub.execute_input":"2025-12-22T03:49:28.030047Z","iopub.status.idle":"2025-12-22T03:49:28.414650Z","shell.execute_reply.started":"2025-12-22T03:49:28.030016Z","shell.execute_reply":"2025-12-22T03:49:28.413672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"! head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-22T03:49:34.925533Z","iopub.execute_input":"2025-12-22T03:49:34.925890Z","iopub.status.idle":"2025-12-22T03:49:36.170363Z","shell.execute_reply.started":"2025-12-22T03:49:34.925860Z","shell.execute_reply":"2025-12-22T03:49:36.168906Z"}},"outputs":[],"execution_count":null}]}