{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-31T05:36:13.049006Z","iopub.execute_input":"2023-05-31T05:36:13.050125Z","iopub.status.idle":"2023-05-31T05:36:13.069420Z","shell.execute_reply.started":"2023-05-31T05:36:13.050086Z","shell.execute_reply":"2023-05-31T05:36:13.068451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Project Overview\n\n# This project leverages machine learning techniques to predict student performance based on game play data. \n# It utilizes the TensorFlow Decision Forests library, TensorFlow Addons, and the TensorFlow library for model \n# development, training, and prediction. \n\n# Data\n# The dataset is sourced from a Kaggle competition, specifically designed for predicting student performance \n# based on their interactions within a learning-based game. The data is rich, including event types, time elapsed, \n# coordinates of the game room and screen, hover duration, and specific text identifiers. The dataset has been \n# appropriately cast into relevant data types to optimize performance and memory usage.\n\n# Data Exploration and Visualization\n# Extensive exploratory data analysis has been conducted on this dataset, including visualizing the distribution \n# of correct and incorrect answers for each question in the dataset. This is an important step that helps in \n# understanding the difficulty of each question for the students. It's done through a series of bar plots, each \n# representing a different question.\n\n# Moreover, the distribution of numerical variables, such as elapsed time, room and screen coordinates, and hover \n# duration, has also been examined. This includes visualization via Kernel Density Estimation plots, pair plots, \n# box plots, and a correlation heatmap. These visualizations provide insights about the distribution, spread, \n# outliers, correlations and relationships within these features.\n\n# Feature Engineering\n# The project employs a function for feature engineering, processing both categorical and numerical variables, \n# grouping them by session ID and level group, and calculating unique counts for categorical features and mean \n# and standard deviation for numerical features. \n\n# Model Development and Training\n# Each question level group is trained with a Gradient Boosted Trees model, a powerful ensemble machine learning \n# model known for its accuracy and efficiency. The model is trained on the engineered dataset, with each session \n# as a separate instance. The training process involves defining the Gradient Boosted Trees Model, compiling it \n# with the accuracy metric, fitting the model on the training data, and storing the model for future use.\n\n# Evaluation and Prediction\n# The trained models are evaluated on a validation set, and the accuracy for each model is stored. These evaluation \n# scores are used to calculate the average accuracy across all questions, providing a holistic view of the model's \n# performance.\n\n# Predictions for each question are made on the validation dataset, and these predictions are stored in a dataframe. \n\n# The project is thus an end-to-end machine learning task, handling data loading, preprocessing, feature engineering, \n# model training, evaluation, and prediction. It illustrates the process of building a student performance prediction \n# system using gameplay data and TensorFlow libraries.\n\n\n# Steps:\n# Step 1: Print the versions of TensorFlow, TensorFlow Addons, and TensorFlow Decision Forests. This is crucial to ensure compatibility between the different libraries used in the code.\n# Step 2: We are setting the data types for each of the columns in our data. This helps pandas interpret the data correctly and can save memory by using more efficient types.\n# Step 3: Load the training data and print the size of the data\n# Step 4: Load labels data. We are extracting the 'session' and 'q' parts of 'session_id' and storing them in separate columns for later use.\n# Step 5: Visualize the distribution of the 'correct' column in the labels data. This is crucial for understanding the distribution of correct and incorrect answers.\n# Step 6: Feature engineering. We group the data by 'session_id' and 'level_group', and then compute the number of unique values for each categorical column, and the mean and standard deviation for each numerical column.\n# Step 7: Splitting the data into a training set and a validation set.\n# Step 8: Training a Gradient Boosted Trees model for each question, evaluate the model, and store the predictions.\n# Step 9: Visualize the standard deviation of numerical variables. This helps us understand the variability of these variables across different level groups.\n# Step 10: Visualize the correlation heatmap of numerical variables. This can help us understand the relationships between different numerical variables.\n# Step 11: Visualize the Kernel Density Estimation (KDE) plots for all the numerical columns. KDE plots can help us visualize the distribution of numerical variables.\n# Step 12: Visualize the pair plot of a subset of numerical variables. This can help us understand the pairwise relationships between these variables.\n# Step 13: Visualize the box plots for 'elapsed_time' and 'hover_duration' variables. Box plots can help us understand the distribution and outliers of these variables.\n\n# Key decisions:\n# 1. We decided to use Gradient Boosted Trees model as it usually provides a good balance between accuracy and model interpretability.\n# 2. We decided to perform feature engineering by grouping the data by 'session_id' and 'level_group', and then compute the number of unique values for each categorical column, and the mean and standard deviation for each numerical column. This could help capture some patterns within each group.\n# 3. We decided to visualize the distribution of numerical variables and the relationships between them using different types of plots. This could help us better understand the data and make better decisions on model selection and feature engineering.\n# 4. We decided to train a separate model for each question. This is because the pattern of students' performance could be different for different questions.","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:45:55.442930Z","iopub.execute_input":"2023-05-31T05:45:55.443377Z","iopub.status.idle":"2023-05-31T05:45:55.450161Z","shell.execute_reply.started":"2023-05-31T05:45:55.443347Z","shell.execute_reply":"2023-05-31T05:45:55.449162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\nimport seaborn as sns  # Added seaborn for improved visualizations\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:44:45.170316Z","iopub.execute_input":"2023-05-31T05:44:45.171752Z","iopub.status.idle":"2023-05-31T05:44:45.177747Z","shell.execute_reply.started":"2023-05-31T05:44:45.171703Z","shell.execute_reply":"2023-05-31T05:44:45.176605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the versions of TensorFlow, TensorFlow Addons, and TensorFlow Decision Forests. \n# This is important to ensure the compatibility of the libraries and to assist with debugging if needed.\n\nprint(\"TensorFlow Decision Forests v\" + tfdf.__version__)\nprint(\"TensorFlow Addons v\" + tfa.__version__)\nprint(\"TensorFlow v\" + tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:36:23.145381Z","iopub.execute_input":"2023-05-31T05:36:23.145970Z","iopub.status.idle":"2023-05-31T05:36:23.152220Z","shell.execute_reply.started":"2023-05-31T05:36:23.145941Z","shell.execute_reply":"2023-05-31T05:36:23.151444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the data types for each column of the dataset. \n# This helps pandas interpret the data correctly and can also save memory by using more efficient data types.\n# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\n\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\n\n# Load the training dataset with the defined data types and print the shape of the dataset to confirm its size.\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:36:23.154767Z","iopub.execute_input":"2023-05-31T05:36:23.155350Z","iopub.status.idle":"2023-05-31T05:38:35.943233Z","shell.execute_reply.started":"2023-05-31T05:36:23.155319Z","shell.execute_reply":"2023-05-31T05:38:35.942445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first few rows of the dataset for a quick preview.\ndataset_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:38:35.944562Z","iopub.execute_input":"2023-05-31T05:38:35.945057Z","iopub.status.idle":"2023-05-31T05:38:35.990826Z","shell.execute_reply.started":"2023-05-31T05:38:35.945028Z","shell.execute_reply":"2023-05-31T05:38:35.989936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the labels data and separate out the 'session' and 'q' parts of 'session_id' for easier handling later.\nlabels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:38:35.992138Z","iopub.execute_input":"2023-05-31T05:38:35.992652Z","iopub.status.idle":"2023-05-31T05:38:37.480472Z","shell.execute_reply.started":"2023-05-31T05:38:35.992623Z","shell.execute_reply":"2023-05-31T05:38:37.479416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first few rows of the labels for a quick preview.\nlabels.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:38:37.481928Z","iopub.execute_input":"2023-05-31T05:38:37.482234Z","iopub.status.idle":"2023-05-31T05:38:37.497932Z","shell.execute_reply.started":"2023-05-31T05:38:37.482208Z","shell.execute_reply":"2023-05-31T05:38:37.496097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Bar chart for label column**","metadata":{}},{"cell_type":"code","source":"# Plot the distribution of the 'correct' column in the labels data. \n# This helps to understand the distribution of correct and incorrect answers.\n\nplt.figure(figsize=(3, 3))\nplot_df = labels.correct.value_counts()\nplot_df.plot(kind=\"bar\", color=['b', 'c'])","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:38:37.500336Z","iopub.execute_input":"2023-05-31T05:38:37.501213Z","iopub.status.idle":"2023-05-31T05:38:37.774630Z","shell.execute_reply.started":"2023-05-31T05:38:37.501181Z","shell.execute_reply":"2023-05-31T05:38:37.773739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a new figure for plotting the bar charts. The figure is set to be 10x20 in size.\n# This size is chosen to ensure that the individual plots are large enough to clearly see the data, but small enough to fit many plots on one figure.\n# Adjust the subplot layout parameters. \n# `hspace` and `wspace` control the amount of height and width reserved for space between subplots,\n# expressed as a fraction of the average axis height/width.\n\nplt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"\\\"Correct\\\" column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    #print(n, str(n))\n    ax = plt.subplot(6, 3, n)\n\n    # Filter df and plot ticker on the new subplot axis\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['b', 'c'])\n    \n    # Chart formatting\n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:45.566105Z","iopub.execute_input":"2023-05-31T05:42:45.566804Z","iopub.status.idle":"2023-05-31T05:42:48.782613Z","shell.execute_reply.started":"2023-05-31T05:42:45.566770Z","shell.execute_reply":"2023-05-31T05:42:48.781313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the categorical and numerical columns in the dataset.\nCATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:48.784085Z","iopub.execute_input":"2023-05-31T05:42:48.784535Z","iopub.status.idle":"2023-05-31T05:42:48.790348Z","shell.execute_reply.started":"2023-05-31T05:42:48.784503Z","shell.execute_reply":"2023-05-31T05:42:48.789156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform feature engineering on the dataset by computing aggregations like the number of unique values, the mean, and standard deviation for each group.\ndef feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df\n\n\n# Apply the feature engineering function to the dataset.\ndataset_df = feature_engineer(dataset_df)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:38:40.319164Z","iopub.execute_input":"2023-05-31T05:38:40.319498Z","iopub.status.idle":"2023-05-31T05:39:26.034028Z","shell.execute_reply.started":"2023-05-31T05:38:40.319470Z","shell.execute_reply":"2023-05-31T05:39:26.032468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first few rows of the prepared dataset for a quick preview.\ndataset_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:39:26.035746Z","iopub.execute_input":"2023-05-31T05:39:26.036071Z","iopub.status.idle":"2023-05-31T05:39:26.075038Z","shell.execute_reply.started":"2023-05-31T05:39:26.036037Z","shell.execute_reply":"2023-05-31T05:39:26.073955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the summary statistics of the prepared dataset.\ndataset_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:39:26.076309Z","iopub.execute_input":"2023-05-31T05:39:26.076660Z","iopub.status.idle":"2023-05-31T05:39:26.236230Z","shell.execute_reply.started":"2023-05-31T05:39:26.076632Z","shell.execute_reply":"2023-05-31T05:39:26.235062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the dataset into a training set and a validation set.\ndef split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset_df.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\n\n# Perform the split and print the number of examples in the training and validation sets.\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:39:26.237602Z","iopub.execute_input":"2023-05-31T05:39:26.237958Z","iopub.status.idle":"2023-05-31T05:39:26.348433Z","shell.execute_reply.started":"2023-05-31T05:39:26.237928Z","shell.execute_reply":"2023-05-31T05:39:26.347199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print all available models in TensorFlow Decision Forests.\ntfdf.keras.get_all_models()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:39:26.349755Z","iopub.execute_input":"2023-05-31T05:39:26.350085Z","iopub.status.idle":"2023-05-31T05:39:26.357225Z","shell.execute_reply.started":"2023-05-31T05:39:26.350059Z","shell.execute_reply":"2023-05-31T05:39:26.356189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nVALID_USER_LIST = valid_x.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}\n","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:39:26.358618Z","iopub.execute_input":"2023-05-31T05:39:26.359469Z","iopub.status.idle":"2023-05-31T05:39:26.369806Z","shell.execute_reply.started":"2023-05-31T05:39:26.359424Z","shell.execute_reply":"2023-05-31T05:39:26.368693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n        \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = train_x.loc[train_x.level_group == grp].copy() # added .copy()\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp].copy() # added .copy()\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # There's one more step required before we can train the model. \n    # We need to convert the datatset from Pandas format (pd.DataFrame)\n    # into TensorFlow Datasets format (tf.data.Dataset).\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n    # We will now create the Gradient Boosted Trees Model with default settings. \n    # By default the model is set to train for a classification task.\n    gbtm = tfdf.keras.GradientBoostedTreesModel(verbose=0)\n    gbtm.compile(metrics=[\"accuracy\"])\n\n    # Train the model.\n    gbtm.fit(x=train_ds)\n\n    # Store the model\n    models[f'{grp}_{q_no}'] = gbtm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    inspector = gbtm.make_inspector()\n    inspector.evaluation()\n    evaluation = gbtm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]         \n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = gbtm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()\n    \n    #NOTE:\n    # The warning does not appear to be causing any errors or unexpected behaviour in the script. \n    # The process of iterating over the questions, creating and training the Gradient Boosted Trees models, \n    # and making predictions all appears to be functioning as expected. \n    # The model is training and making predictions, and the accuracy of the model is being printed to the console.","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:39:26.371640Z","iopub.execute_input":"2023-05-31T05:39:26.372266Z","iopub.status.idle":"2023-05-31T05:42:15.725508Z","shell.execute_reply.started":"2023-05-31T05:39:26.372222Z","shell.execute_reply":"2023-05-31T05:42:15.724468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the distribution of the 'correct' column in the labels data. This helps to understand the distribution of correct and incorrect answers.\n# In other words, it gives us an idea about how balanced or imbalanced the data is.\n# An imbalanced dataset could bias the model towards the majority class, so it's an important aspect to consider.\n\nfor name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:15.727097Z","iopub.execute_input":"2023-05-31T05:42:15.728377Z","iopub.status.idle":"2023-05-31T05:42:15.734677Z","shell.execute_reply.started":"2023-05-31T05:42:15.728340Z","shell.execute_reply":"2023-05-31T05:42:15.733677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the standard deviation of numerical variables by level group.\n# This visualization helps to understand how the variability of these variables changes across different level groups.\n# If a particular level shows more variability, that could imply the presence of more diverse data patterns within that level, which might affect model performance.\n\nfig, axs = plt.subplots(3, 2, figsize=(10, 10))\nfor name, data in dataset_df.groupby('level_group'):\n    axs[0, 0].plot(range(1, len(data['room_coor_x_std'])+1), data['room_coor_x_std'], label=name)\n    axs[0, 1].plot(range(1, len(data['room_coor_y_std'])+1), data['room_coor_y_std'], label=name)\n    axs[1, 0].plot(range(1, len(data['screen_coor_x_std'])+1), data['screen_coor_x_std'], label=name)\n    axs[1, 1].plot(range(1, len(data['screen_coor_y_std'])+1), data['screen_coor_y_std'], label=name)\n    axs[2, 0].plot(range(1, len(data['hover_duration_std'])+1), data['hover_duration_std'], label=name)\n    axs[2, 1].plot(range(1, len(data['elapsed_time_std'])+1), data['elapsed_time_std'], label=name)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:15.736167Z","iopub.execute_input":"2023-05-31T05:42:15.736994Z","iopub.status.idle":"2023-05-31T05:42:17.734315Z","shell.execute_reply.started":"2023-05-31T05:42:15.736960Z","shell.execute_reply":"2023-05-31T05:42:17.731553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the correlation heatmap of numerical variables.\n# This visualization helps to identify the strength and direction of the relationship between different numerical variables.\n# If two variables are highly correlated, they might carry similar information. In some cases, one might be removed to reduce redundancy.\n\ncorrelation = dataset_df[NUMERICAL].corr()\nplt.figure(figsize=(10, 8))\nsns.heatmap(correlation, annot=True, cmap=\"YlGnBu\")\nplt.title(\"Correlation heatmap of numerical variables\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:17.735895Z","iopub.execute_input":"2023-05-31T05:42:17.736918Z","iopub.status.idle":"2023-05-31T05:42:18.287753Z","shell.execute_reply.started":"2023-05-31T05:42:17.736877Z","shell.execute_reply":"2023-05-31T05:42:18.286407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kernel Density Estimation (KDE) plots for all the numerical columns in the provided dataset, which can help visualize the shape, central tendency, and spread of the numerical variables.\n# A KDE plot is a method of visualizing the distribution of observations in a dataset, similar to a histogram. \n# KDE plots can be more useful than histograms because they provide a smooth curve, which can be easier to understand in some cases.\n# Provides an understanding of the distributions of the numerical columns in the DataFrame. \n# This helps with exploratory data analysis by providing insights about the range, skewness, or any anomalies like outliers in the data.\n\ndef plot_distribution(df, column, title):\n    plt.figure(figsize=(10, 5))\n    sns.kdeplot(data=df, x=column, fill=True)\n    plt.title(title)\n    plt.show()\n\n# Visualize distributions of numerical features with respect to the target\nfor col in NUMERICAL:\n    plot_distribution(dataset_df, col, f\"Distribution of {col}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:18.289274Z","iopub.execute_input":"2023-05-31T05:42:18.289628Z","iopub.status.idle":"2023-05-31T05:42:22.353635Z","shell.execute_reply.started":"2023-05-31T05:42:18.289596Z","shell.execute_reply":"2023-05-31T05:42:22.352294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize pairs of numerical variables using pair plots.\n# Pair plots provide a bird’s eye view of the relationships among numerical variables, allowing us to see correlations and outlier patterns.\n# They can also show how the variables are distributed individually, which provides additional context.\n\nNUMERICAL_SUBSET = ['elapsed_time', 'room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\nsns.pairplot(dataset_df[NUMERICAL_SUBSET], diag_kind=\"kde\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:22.354955Z","iopub.execute_input":"2023-05-31T05:42:22.355693Z","iopub.status.idle":"2023-05-31T05:42:42.280629Z","shell.execute_reply.started":"2023-05-31T05:42:22.355661Z","shell.execute_reply":"2023-05-31T05:42:42.279693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the distribution of 'elapsed_time' and 'hover_duration' across different level groups using box plots.\n# Box plots are a great way to understand the statistical summary (like median, quartiles, outliers) of these variables across different levels.\n# Understanding these distributions might hint at how these features could affect a student's performance.\n\nplt.figure(figsize=(15, 5))\n\n# Box plot for 'elapsed_time'\nplt.subplot(1, 2, 1)\nsns.boxplot(x='level_group', y='elapsed_time', data=dataset_df)\nplt.title('Box plot of elapsed_time by level_group')\n\n# Box plot for 'hover_duration'\nplt.subplot(1, 2, 2)\nsns.boxplot(x='level_group', y='hover_duration', data=dataset_df)\nplt.title('Box plot of hover_duration by level_group')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:42:42.281887Z","iopub.execute_input":"2023-05-31T05:42:42.282779Z","iopub.status.idle":"2023-05-31T05:42:42.814851Z","shell.execute_reply.started":"2023-05-31T05:42:42.282747Z","shell.execute_reply":"2023-05-31T05:42:42.813765Z"},"trusted":true},"execution_count":null,"outputs":[]}]}