{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Set up environment ","metadata":{}},{"cell_type":"markdown","source":"## Import library","metadata":{}},{"cell_type":"code","source":"from IPython.display import display, clear_output\n\nimport pandas as pd\nimport polars as pl\nimport numpy as np\nimport gc\nimport math\nimport polars.selectors as cs\nimport plotly.express as px\nimport plotly.graph_objs as go\nimport plotly.subplots as sp\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.figure_factory as ff\nfrom plotly.offline import init_notebook_mode\nimport matplotlib.image as mpimg\nimport warnings\n# Initialize Plotly for offline use\ninit_notebook_mode(connected=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:38:19.608161Z","iopub.execute_input":"2024-12-08T10:38:19.608445Z","iopub.status.idle":"2024-12-08T10:38:23.492163Z","shell.execute_reply.started":"2024-12-08T10:38:19.608417Z","shell.execute_reply":"2024-12-08T10:38:23.491415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Defining seed and the template for plots\nseed = 42\nplotly_template = 'simple_white'\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:38:23.493657Z","iopub.execute_input":"2024-12-08T10:38:23.493946Z","iopub.status.idle":"2024-12-08T10:38:23.498284Z","shell.execute_reply.started":"2024-12-08T10:38:23.493917Z","shell.execute_reply":"2024-12-08T10:38:23.497263Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import Dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntest_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:38:23.499321Z","iopub.execute_input":"2024-12-08T10:38:23.499536Z","iopub.status.idle":"2024-12-08T10:40:01.047161Z","shell.execute_reply.started":"2024-12-08T10:38:23.499514Z","shell.execute_reply":"2024-12-08T10:40:01.046139Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data discovery process","metadata":{}},{"cell_type":"markdown","source":"## Dataset information","metadata":{}},{"cell_type":"markdown","source":"Features in 'train.csv' and 'test.csv' and present as below:","metadata":{}},{"cell_type":"markdown","source":"| Order | Column Name      | Description                                                                                       |\n|-------|------------------|---------------------------------------------------------------------------------------------------|\n| 1     | session_id       | The ID of the session the event took place in                                                     |\n| 2     | index            | The index of the event for the session                                                            |\n| 3     | elapsed_time     | How much time has passed (in milliseconds) between the start of the session and when the event was recorded |\n| 4     | event_name       | The name of the event type                                                                        |\n| 5     | name             | The event name (e.g. identifies whether a notebook_click is opening or closing the notebook)      |\n| 6     | level            | What level of the game the event occurred in (0 to 22)                                            |\n| 7     | page             | The page number of the event (only for notebook-related events)                                   |\n| 8     | room_coor_x      | The coordinates of the click in reference to the in-game room (only for click events)             |\n| 9     | room_coor_y      | The coordinates of the click in reference to the in-game room (only for click events)             |\n| 10    | screen_coor_x    | The coordinates of the click in reference to the player’s screen (only for click events)          |\n| 11    | screen_coor_y    | The coordinates of the click in reference to the player’s screen (only for click events)          |\n| 12    | hover_duration   | How long (in milliseconds) the hover happened for (only for hover events)                         |\n| 13    | text             | The text the player sees during this event                                                        |\n| 14    | fqid             | The fully qualified ID of the event                                                               |\n| 15    | room_fqid        | The fully qualified ID of the room the event took place in                                        |\n| 16    | text_fqid        | The fully qualified ID of the text                                                                |\n| 17    | fullscreen       | Whether the player is in fullscreen mode                                                          |\n| 18    | hq               | Whether the game is in high-quality                                                               |\n| 19    | music            | Whether the game music is on or off                                                               |\n| 20    | level_group      | Which group of levels - and group of questions - this row belongs to (0-4, 5-12, 13-22)           |\n","metadata":{}},{"cell_type":"markdown","source":"## Train - Test dataset:","metadata":{}},{"cell_type":"markdown","source":"### Quick overview the dataset\n","metadata":{}},{"cell_type":"markdown","source":"I created a function called `dataframe_description` to help me get a quick overview of a DataFrame (`df`). This function prints out important information about the dataset, such as its shape, missing data, duplicates, data types, and feature classifications.\r\n data.\r\n\r\nThis function helps me quickly understand the structure and characteristics of my dataset.","metadata":{}},{"cell_type":"code","source":"def dataframe_description(df):\n    \"\"\"\n    This function prints some basic info about the dataset, including \n    its shape, missing data, duplicates, data types, and feature classifications.\n    \"\"\"\n    # Classify features\n    categorical_features = [col for col in df.columns if df[col].dtype == object]\n    binary_features = [col for col in df.columns if df[col].nunique() <= 2 and df[col].dtype != object]\n    continuous_features = [col for col in df.columns if col not in categorical_features + binary_features]\n    feature_sets = {\n        \"Categorical Features\": categorical_features,\n        \"Continuous Features\": continuous_features,\n        \"Binary Features\": binary_features,\n    }\n\n    # Print dataset shape\n    print(f\"\\n\\033[1m{type(df).__name__} shape\\033[0m: {df.shape}\")\n    print(f\"{df.shape[0]:,} samples - {df.shape[1]:,} attributes\")\n    print(f'\\n\\033[1mDuplicates\\033[0m: {df.duplicated().sum()}\\n')\n\n\n    # Generate tables for each feature set\n    for feature_type in feature_sets.keys():\n        # Create a table for the current feature set\n        if feature_type:\n            print(f'\\n\\033[1m {feature_type}\\033[0m:')\n            print(feature_sets[feature_type])\n        else:\n            print(f\"\\n\\033[1m{feature_type}\\033[0m: No features available.\")\n\n    print(f'\\n\\033[1m{type(df).__name__} Head\\033[0m:\\n')\n    display(df.head())\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:40:01.049364Z","iopub.execute_input":"2024-12-08T10:40:01.049665Z","iopub.status.idle":"2024-12-08T10:40:01.143939Z","shell.execute_reply.started":"2024-12-08T10:40:01.049639Z","shell.execute_reply":"2024-12-08T10:40:01.143034Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataframe_description(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:40:01.145168Z","iopub.execute_input":"2024-12-08T10:40:01.145602Z","iopub.status.idle":"2024-12-08T10:41:00.151135Z","shell.execute_reply.started":"2024-12-08T10:40:01.145541Z","shell.execute_reply":"2024-12-08T10:41:00.150201Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataframe_description(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:41:00.152425Z","iopub.execute_input":"2024-12-08T10:41:00.153095Z","iopub.status.idle":"2024-12-08T10:41:00.187878Z","shell.execute_reply.started":"2024-12-08T10:41:00.153051Z","shell.execute_reply":"2024-12-08T10:41:00.186838Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Descriptive Statistics","metadata":{}},{"cell_type":"markdown","source":"Let's generate a formatted summary of the DataFrame, starts with some descriptive statistics for the DataFrame, such as the mean, standard deviation, minimum, and maximum values for each column.","metadata":{}},{"cell_type":"code","source":"desc = test_df.describe().T.sort_values(by='std' , ascending = False)\nformatted_desc = desc.style.format(\"{:.2f}\").set_table_styles(\n    [{'selector': 'th', 'props': [('font-size', '12pt')]}]\n)\nformatted_desc","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:41:00.188785Z","iopub.execute_input":"2024-12-08T10:41:00.189009Z","iopub.status.idle":"2024-12-08T10:41:00.287906Z","shell.execute_reply.started":"2024-12-08T10:41:00.188969Z","shell.execute_reply":"2024-12-08T10:41:00.287129Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"desc = train_df.describe().T.sort_values(by='std' , ascending = False)\nformatted_desc = desc.style.format(\"{:.2f}\").set_table_styles(\n    [{'selector': 'th', 'props': [('font-size', '12pt')]}]\n)\nformatted_desc","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:41:00.289474Z","iopub.execute_input":"2024-12-08T10:41:00.290049Z","iopub.status.idle":"2024-12-08T10:41:11.333133Z","shell.execute_reply.started":"2024-12-08T10:41:00.289999Z","shell.execute_reply":"2024-12-08T10:41:11.332180Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Number of unique values","metadata":{}},{"cell_type":"code","source":"stat = pd.DataFrame([train_df.nunique(), test_df.nunique(),train_df.isna().sum(), test_df.isna().sum()]).T.fillna(0)\nstat.columns = ['Number of unique values in train', 'Number of unique values in test','Number of Nan values in train', 'Number of Nan values in test']\nstat = stat.sort_values('Number of unique values in train', ascending = False)\nstat.head(30).style.format(\"{:,.0f}\").background_gradient(cmap='YlGn')","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:41:11.334269Z","iopub.execute_input":"2024-12-08T10:41:11.334525Z","iopub.status.idle":"2024-12-08T10:41:36.246923Z","shell.execute_reply.started":"2024-12-08T10:41:11.334498Z","shell.execute_reply":"2024-12-08T10:41:36.246073Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Column-wise Null Value Distribution","metadata":{}},{"cell_type":"code","source":"sorted_stat = pd.DataFrame()\nsorted_stat['Train Percentage'] = stat['Number of Nan values in train'] / len(train_df) * 100\nsorted_stat['Test Percentage'] = stat['Number of Nan values in test'] / len(test_df) * 100\n\n# Initialize the matplotlib figure\nfig, axes = plt.subplots(1, 2, figsize=(14, 8), sharey=True)\n\n# Plot for Train Data\ntrain_plot = sns.barplot(\n    x='Train Percentage',\n    y=sorted_stat.index,\n    data=sorted_stat,\n    palette='YlGn',\n    ax=axes[0]\n)\naxes[0].set_title(\"Train Data\")\naxes[0].set_xlabel(\"Missing Values (%)\")\naxes[0].set_ylabel(\"\")\naxes[0].grid(axis='x', linestyle='--', alpha=0.7)\n\n# Add percentage labels to bars for Train Data\nfor bar in train_plot.patches:\n    width = bar.get_width()\n    axes[0].text(\n        width + 0.5,  \n        bar.get_y() + bar.get_height() / 2,  \n        f\"{width:.2f}%\",\n        ha='left',\n        va='center',\n        fontsize=10\n    )\n\n# Plot for Test Data\ntest_plot = sns.barplot(\n    x='Test Percentage',\n    y=sorted_stat.index,\n    data=sorted_stat,\n    palette='YlGn',\n    ax=axes[1]\n)\naxes[1].set_title(\"Test Data\")\naxes[1].set_xlabel(\"Missing Values (%)\")\naxes[1].set_ylabel(\"\")\naxes[1].grid(axis='x', linestyle='--', alpha=0.7)\n\n# Add percentage labels to bars for Test Data\nfor bar in test_plot.patches:\n    width = bar.get_width()\n    axes[1].text(\n        width + 0.5,\n        bar.get_y() + bar.get_height() / 2,\n        f\"{width:.2f}%\",\n        ha='left',\n        va='center',\n        fontsize=10\n    )\n\n# Update overall title\nfig.suptitle(\"Column-wise Null Value Distribution\", fontsize=16)\nfig.tight_layout(rect=[0, 0, 1, 0.95])\n\nplt.show()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:41:36.250858Z","iopub.execute_input":"2024-12-08T10:41:36.251161Z","iopub.status.idle":"2024-12-08T10:41:37.283419Z","shell.execute_reply.started":"2024-12-08T10:41:36.251132Z","shell.execute_reply":"2024-12-08T10:41:37.282609Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p style=\"font-size: 17px; font-weight: bold;\">\n    Key observations:\n</p> \n<p style=\"font-size: 15px;\">\n    <b>1. The training data has significantly higher missing values for the columns \"hover_duration\" (92.39%), \"page\" (97.85%), \"text_fqid\" (63.43%), \"text\" (63.43%) and \"fqid\" (31.47%) compared to the rest of the columns.</b><br>\n    <b>2. The test data has high missing values for the columns \"hover_duration\" (90.53%), \"page\" (95.90%), \"text_fqid\" (68.83%), \"text\" (68.83%) and \"fqid\" (31.47%) compared to the rest of the columns..</b><br>\n    <b>3. The remaining columns in both datasets have very low or no missing values, indicating they are well-populated.</b><br>\n    <b>4. The null value distributions are consistent between the training and test data, suggesting the datasets have similar characteristics in terms of missing data.</b>\n</p>","metadata":{}},{"cell_type":"markdown","source":"### Strategy","metadata":{}},{"cell_type":"markdown","source":"XGBoost, CatBoost, and LightGBM are powerful gradient boosting algorithms that handle missing values effectively, each in their unique way.\n\nXGBoost supports missing values by default. During training, it learns the best direction to take when encountering a missing value in a feature. This means that the model can decide whether to treat the missing value as a zero or to follow a different path in the decision tree.\n\nCatBoost also handles missing values natively. It treats missing values as a separate category and processes them accordingly. CatBoost can handle missing values in both numerical and categorical features without requiring any imputation. It uses a special algorithm to process missing values, ensuring that they are treated correctly during training.\n\nLightGBM handles missing values by default as well. It uses NA (NaN) to represent missing values. During training, LightGBM decides the best way to handle missing values by allocating them to the side of the split that reduces the loss the most. This allows the model to effectively manage missing data without the need for imputation.\n\nIn my best effort, I will try to use the aboves model to solved the prolem.","metadata":{}},{"cell_type":"markdown","source":"### Feature Correlation Heatmap","metadata":{}},{"cell_type":"code","source":"def plot_correlation(df):\n    '''\n    This function is responsible for plotting a correlation map among features in the dataset\n    '''\n    # Select continuous columns and make a copy\n    binary_features = [col for col in df.columns if df[col].nunique() <= 2 and df[col].dtype != object]\n    numeric_df = df.select_dtypes(include='number').drop(columns=binary_features).copy()\n    \n    # Calculate the correlation matrix\n    corr = np.round(numeric_df.corr(), 2)\n    \n    # Create a mask for the upper triangle\n    mask = np.triu(np.ones_like(corr, dtype=bool))\n    \n    # Set up the matplotlib figure\n    plt.figure(figsize=(12, 8))\n    \n    # Draw the heatmap with the mask and correct aspect ratio\n    sns.heatmap(corr, mask=mask, annot=True, fmt=\".2f\", cmap='YlGn', cbar_kws={\"shrink\": .8}, linewidths=.5)\n    \n    # Add title and adjust layout\n    plt.title('Feature Correlation Heatmap', fontsize=18, fontweight='bold')\n    plt.xlabel('Features', fontsize=14)\n    plt.ylabel('Features', fontsize=14)\n    plt.xticks(rotation=45)\n    plt.yticks(rotation=0)\n    plt.tight_layout()\n    \n    # Show the plot\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:41:37.284366Z","iopub.execute_input":"2024-12-08T10:41:37.284639Z","iopub.status.idle":"2024-12-08T10:41:37.291876Z","shell.execute_reply.started":"2024-12-08T10:41:37.284609Z","shell.execute_reply":"2024-12-08T10:41:37.290991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This function, plot_correlation, is designed to create a visual representation of the relationships between different features in a dataset. Through this, we can get a clear and visually appealing way to understand the relationships between the features in the dataset.","metadata":{}},{"cell_type":"code","source":"plot_correlation(train_df)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:41:37.292934Z","iopub.execute_input":"2024-12-08T10:41:37.293277Z","iopub.status.idle":"2024-12-08T10:42:04.289884Z","shell.execute_reply.started":"2024-12-08T10:41:37.293221Z","shell.execute_reply":"2024-12-08T10:42:04.289090Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p style=\"font-size: 17px; font-weight: bold;\">\n    Key observations:\n</p>\n\n1. **Highest Positive Correlation**: The features \"page\" and \"level\" have the highest positive correlation with a value of 0.95. This suggests that as the level increases, the page number also increases.\n2. **Strong Positive Correlation**: \"screen_coor_x\" anroom_coor_xr_y\" have a strong positive correlation of 0.69, indicating that these coordinates tend to increase together.\n3. **Strong Negative Correlation**: \"screen_coor_y\" and \"room_coor_y\" have a strong negative correlation of -0.77, meaning that as one increases, the other decreases.\n4. **Low or Negligible Correlations**: Most other features have low or negligible correlations with each other","metadata":{}},{"cell_type":"code","source":"plot_correlation(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:42:04.291415Z","iopub.execute_input":"2024-12-08T10:42:04.291778Z","iopub.status.idle":"2024-12-08T10:42:04.764289Z","shell.execute_reply.started":"2024-12-08T10:42:04.291738Z","shell.execute_reply":"2024-12-08T10:42:04.763255Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<p style=\"font-size: 17px; font-weight: bold;\">\n    Key observations:\n</p>\n\n1. **Highest Positive Correlation**: The features (\"page\" - \"level\": 0.99), (\"index\" - \"level\": 0.92), (\"page\" - \"index\": 0.91)  have the highest positive correlations. This suggests that the three feature may have hidden relationship.\n2. **Strong Positive Correlation**: (\"session_level\" - \"session_id\": 0.82), (\"screen_coor_x\" - \"room_coor_x\": 0.74), (\"elapsed_time\" - \"index\": 0.61) have a strong positive correlation of 0.80, indicating that these features tend to increase together.\n3. **Strong Negative Correlation**: \"screen_coor_y\" and \"room_coor_y\" have a strong negative correlation of -0.80, meaning that as one increases, the other decreases.\n4. **Low or Negligible Correlations**: Most other features have low or negligible correlations with each other, as indicated by values close to 0.\n","metadata":{}},{"cell_type":"markdown","source":"><p style=\"font-size: 20px; font-weight: bold;\">\n    Summary:\n</p>\n\n>\n>1. **Strong positive correlations** (e.g., \"page\" and \"level\" or \"screen_coor_x\" and \"room_coor_x\") suggest interdependence and potential hierarchical relationships in the data.\n>  \n>2. **Strong negative correlations** (e.g., \"screen_coor_y\" and \"room_coor_y\") indicate inverse or mirrored relationships, particularly in spatial features.\n> \n>3. **Low or negligible correlations** imply independence or weak linear relationships, requiring further analysis for non-linear dependencies.\n>\n>4. **Differences in correlation** strengths between train and test datasets highlight potential variations that should be addressed for model generalization.  ","metadata":{}},{"cell_type":"markdown","source":"### Discussion:\n\n- I want to analyze categorical and numerical data separately. \n- Can I assume that categorical data is covered by train and test? If not, it needs to be supplemented with null.\n  \n- Are the users to be predicted unique?\n- Should features be created for each group, and q_id in(0,1,2,3,4) be predicted using features 0-4, 5-12, 13-22?","metadata":{}},{"cell_type":"code","source":"def plot_distplot(df, x):\n    '''\n    This function creates a distribution plot for continuous variables using Seaborn.\n    '''\n    feature = df[x]\n    \n    plt.figure(figsize=(10, 6))\n    sns.kdeplot(feature, fill=True, color='blue', alpha=0.5)\n    \n    plt.title(f'Distribution Plot\\n{x}', fontsize=16, fontweight='bold', loc='left')\n    plt.xlabel(x, fontsize=12)\n    plt.ylabel('Density', fontsize=12)\n    plt.grid(visible=True, linestyle='--', alpha=0.6)\n    \n    plt.tight_layout()\n    plt.show()\n    \n    \ndef boxplot(df, y, x, hue=None):\n    '''\n    This function plots a boxplot of Y versus X using Seaborn.\n    '''\n    plt.figure(figsize=(10, 6))\n    sns.boxplot(data=df, x=x, y=y, hue=hue, palette=\"Set2\")\n    \n    plt.title(f'Boxplot\\n{y} by {x}', fontsize=16, fontweight='bold', loc='left')\n    plt.xlabel(x, fontsize=12)\n    plt.ylabel(y, fontsize=12)\n    plt.xticks(rotation=-45)  # Rotate x-axis labels if needed\n    plt.tight_layout()\n    plt.show()\n    \n    \ndef barplot(df, feat, n_top=None):\n    '''\n    This function organizes the top n value counts of any attribute and plots a barplot using Seaborn.\n    '''\n    counts = df[feat].value_counts()\n    \n    # Select top n values if specified\n    if n_top:\n        counts = counts.head(n_top)\n    \n    plt.figure(figsize=(10, 6))\n    sns.barplot(x=counts.index, y=counts.values, palette=\"YlGnBu\")\n    \n    # Add value annotations\n    for index, value in enumerate(counts.values):\n        plt.text(index, value, str(value), ha='center', va='bottom', fontsize=10)\n    \n    plt.title(f'Frequency of values in {feat}', fontsize=16, fontweight='bold', loc='left')\n    plt.xlabel(feat, fontsize=12)\n    plt.ylabel('Count', fontsize=12)\n    plt.xticks(rotation=-45)  # Rotate x-axis labels if necessary\n    plt.tight_layout()\n    plt.show()\n    \ndef plot_histogram_matrix(df):\n    '''\n    This function identifies all continuous features within the dataset and plots\n    a matrix of histograms for each attribute using Seaborn.\n    '''\n    # Identify continuous features\n    continuous_features = [\n        feat for feat in df.columns\n        if df[feat].nunique() > 2 and df[feat].dtype != object\n    ]\n\n    num_cols = 2\n    num_rows = math.ceil(len(continuous_features) / num_cols)\n\n    # Set up the figure\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 5 * num_rows))\n    axes = axes.flatten()  # Flatten for easier iteration\n\n    for i, feature in enumerate(continuous_features):\n        sns.histplot(\n            data=df,\n            x=feature,\n            kde=False,\n            ax=axes[i]\n        )\n        axes[i].set_title(feature)\n        axes[i].set_xlabel('Value')\n        axes[i].set_ylabel('Frequency')\n\n    # Turn off unused subplots\n    for j in range(len(continuous_features), len(axes)):\n        axes[j].axis('off')\n\n    # Add an overall title\n    plt.suptitle('Histogram Matrix of Continuous Features', fontsize=16, y=1.02)\n    plt.tight_layout()\n    plt.show()\n    \ndef plot_boxplot_matrix(df):\n    '''\n    This function identifies all continuous features within the dataset and plots\n    a matrix of boxplots for each attribute using Seaborn.\n    '''\n    \n    # Identify continuous features (those with more than 2 unique values and non-object dtype)\n    continuous_features = [feat for feat in df.columns if df[feat].nunique() > 2 and df[feat].dtype != 'object']\n    \n    num_cols = 2\n    num_rows = (len(continuous_features) + 1) // num_cols  # Calculate number of rows\n    \n    # Create the subplots grid\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(12, num_rows * 6))\n    axes = axes.flatten()  # Flatten axes to easily iterate over them\n\n    # Loop through features and create boxplots\n    for i, feature in enumerate(continuous_features):\n        sns.boxplot(data=df, x=feature, ax=axes[i], palette=\"Set2\")\n        axes[i].set_title(f'Boxplot: {feature}', fontsize=12, fontweight='bold')\n        axes[i].set_xlabel('')\n        axes[i].set_ylabel('')\n        axes[i].tick_params(axis='x', rotation=-45)  # Rotate x-axis labels for better readability\n    \n    # Adjust layout\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:42:04.765591Z","iopub.execute_input":"2024-12-08T10:42:04.765883Z","iopub.status.idle":"2024-12-08T10:42:04.782402Z","shell.execute_reply.started":"2024-12-08T10:42:04.765855Z","shell.execute_reply":"2024-12-08T10:42:04.781550Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Categorical data","metadata":{}},{"cell_type":"markdown","source":"This code snippet is designed to check the frequency distribution for each categorical feature in a DataFrame. This approach helps to quickly understand the distribution of values in each categorical feature, providing insights into the dataset's structure and potential patterns.","metadata":{}},{"cell_type":"code","source":"# Checking the frequency distribution for each categorical feature\ncategorical_columns = ['event_name', 'name', 'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']\n\nfor col in categorical_columns:\n    print(f'Value counts for {col} in train_df:')\n    print(train_df[col].value_counts())\n    print('\\n')\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:42:04.783660Z","iopub.execute_input":"2024-12-08T10:42:04.783988Z","iopub.status.idle":"2024-12-08T10:42:15.559428Z","shell.execute_reply.started":"2024-12-08T10:42:04.783962Z","shell.execute_reply":"2024-12-08T10:42:15.558512Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Do the same with `test_df`","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['event_name', 'name', 'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']\n\nfor col in categorical_columns:\n    print(f'Value counts for {col} in test_df:')\n    print(test_df[col].value_counts())\n    print('\\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:42:15.560515Z","iopub.execute_input":"2024-12-08T10:42:15.560808Z","iopub.status.idle":"2024-12-08T10:42:15.574854Z","shell.execute_reply.started":"2024-12-08T10:42:15.560783Z","shell.execute_reply":"2024-12-08T10:42:15.574023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = ['event_name', 'name', 'level_group']\n\nplt.figure(figsize=(12, 16))\n\n# Plot for train_df\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(3, 2, 2*i-1)\n    sns.countplot(data=train_df, x=col, palette='Set3')\n    plt.title(f'Distribution of {col} in train_df')\n    plt.xticks(rotation=45)\n\n# Plot for test_df\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(3, 2, 2*i)\n    sns.countplot(data=test_df, x=col, palette='Set3')\n    plt.title(f'Distribution of {col} in test_df')\n    plt.xticks(rotation=45)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:42:15.575920Z","iopub.execute_input":"2024-12-08T10:42:15.576140Z","iopub.status.idle":"2024-12-08T10:42:54.385016Z","shell.execute_reply.started":"2024-12-08T10:42:15.576117Z","shell.execute_reply":"2024-12-08T10:42:54.384115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> **Observation:**\n> \n> - The distributions between training and test datasets are relatively consistent, which is advantageous for model training and evaluation.\n> \n> - The dominance of a few event types and name categories indicates potential focus areas for analysis and modeling..","metadata":{}},{"cell_type":"code","source":"categorical_columns = ['event_name', 'name', 'level_group']\n\nplt.figure(figsize=(12, 16))\n\n# Plot for train_df\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(3, 2, 2*i-1)\n    train_df[col].value_counts().plot.pie(autopct='%1.1f%%', colors=sns.color_palette('Set3', len(train_df[col].unique())))\n    plt.title(f'Proportion of {col} in train_df')\n    plt.ylabel('')\n\n# Plot for test_df\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(3, 2, 2*i)\n    test_df[col].value_counts().plot.pie(autopct='%1.1f%%', colors=sns.color_palette('Set3', len(test_df[col].unique())))\n    plt.title(f'Proportion of {col} in test_df')\n    plt.ylabel('')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:42:54.386310Z","iopub.execute_input":"2024-12-08T10:42:54.386760Z","iopub.status.idle":"2024-12-08T10:43:04.268687Z","shell.execute_reply.started":"2024-12-08T10:42:54.386714Z","shell.execute_reply":"2024-12-08T10:43:04.267536Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> **Observation:**\n> \n> - The user engagement patterns in both datasets are consistent.\n","metadata":{}},{"cell_type":"markdown","source":"General coordinates statistics make little or no sense. We need click density for each specific room:","metadata":{}},{"cell_type":"code","source":"rooms = train_df['room_fqid'].unique()\nfor room in rooms:\n    data_room = train_df[train_df['room_fqid'] == room]\n    data_room = data_room[['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y']].dropna().reset_index(drop=True)\n    data_room = data_room.apply(pd.to_numeric, errors='coerce')\n    g = sns.pairplot(\n        data=data_room,\n        x_vars=[\"room_coor_x\"],\n        y_vars=[\"room_coor_y\"],\n        kind='hist',\n        height=6\n    )\n    g.fig.suptitle(f\"{room}: game room coordinates\")\n    g.savefig('g0.png', dpi=300)\n    plt.close(g.fig)\n    \n    g = sns.pairplot(\n        data=data_room,\n        x_vars=[\"screen_coor_x\"],\n        y_vars=[\"screen_coor_y\"],\n        kind='hist',\n        height=6\n    )\n    g.fig.suptitle(f\"{room}: player's screen coordinates\")\n    g.savefig('g1.png', dpi=300)\n    plt.close(g.fig)\n    \n    f, ax = plt.subplots(1, 2, figsize=(20, 20))\n    ax[0].imshow(mpimg.imread('g0.png'))\n    ax[1].imshow(mpimg.imread('g1.png'))\n    [axarr.set_axis_off() for axarr in ax.ravel()]\n    plt.tight_layout()\n    plt.show()\n\ndel room, rooms, data_room, g, f, ax","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:43:04.269852Z","iopub.execute_input":"2024-12-08T10:43:04.270120Z","iopub.status.idle":"2024-12-08T10:45:07.736038Z","shell.execute_reply.started":"2024-12-08T10:43:04.270094Z","shell.execute_reply":"2024-12-08T10:45:07.735105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"A high density of clicks shows the locations of important objects or characters. Now let's review the geo-location path for a randomly selected session:","metadata":{}},{"cell_type":"code","source":"def plot_geo_location(df, session_id):\n    session_df = df[df['session_id'] == session_id]\n    rooms = session_df['room_fqid'].unique()\n    for room in rooms:\n        session_df_room = session_df[session_df['room_fqid'] == room]\n        session_df_room = session_df_room[['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y']].dropna().reset_index(drop=True)        \n        plt.figure(figsize=(15, 6))        \n        \n        # room\n        x = session_df_room['room_coor_x']\n        y = session_df_room['room_coor_y']\n        plt.subplot(1, 5, (1, 3))\n        plt.plot(x, y, zorder=0, lw=0.5, color='steelblue')\n        plt.scatter(x, y, s=5, color='grey')\n        plt.scatter(x[0], y[0], s=200, lw=5, color='gold', marker='*')\n        plt.scatter(x[-1:], y[-1:], s=200, lw=5, color='crimson', marker='*')\n        plt.title(f\"{session_id}: {room} (room)\")\n        plt.legend(['Cursor path', 'Click position', 'Start', 'End'])\n        plt.gca().set_aspect('equal', adjustable='box')\n        plt.xlim(-2000, 1300)\n        plt.ylim(-920, 550)\n        plt.xlabel(\"room_coor_x\")\n        plt.ylabel(\"room_coor_y\")      \n    \n        # screen\n        x = session_df_room['screen_coor_x']\n        y = session_df_room['screen_coor_y']\n        plt.subplot(1, 5, (4, 5))\n        plt.plot(x, y, zorder=0, lw=0.5, color='lightcoral')\n        plt.scatter(x, y, s=5, color='grey')\n        plt.scatter(x[0], y[0], s=200, lw=5, color='gold', marker='*')\n        plt.scatter(x[-1:], y[-1:], s=200, lw=5, color='crimson', marker='*')\n        plt.title(f\"session {session_id}: {room} (screen)\")\n        plt.legend(['Cursor path', 'Click position', 'Start', 'End'])\n        plt.gca().set_aspect('equal', adjustable='box')\n        plt.xlabel(\"screen_coor_x\")\n        plt.ylabel(\"screen_coor_y\")\n        plt.xlim(0, 2000)\n        plt.ylim(0, 1500)\n        plt.tight_layout()\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:07.737382Z","iopub.execute_input":"2024-12-08T10:45:07.738101Z","iopub.status.idle":"2024-12-08T10:45:07.754017Z","shell.execute_reply.started":"2024-12-08T10:45:07.738054Z","shell.execute_reply":"2024-12-08T10:45:07.753156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"idx_dupl = train_df[['session_id', 'index']]\nidx_dupl = idx_dupl[idx_dupl.duplicated()]\nsession_ids = train_df['session_id'].unique()\nsession_ids = session_ids[np.isin(session_ids, idx_dupl['session_id'].unique(), invert=True)]\ndel idx_dupl\ngc.collect()\nplot_geo_location(train_df, np.random.choice(session_ids))\ndel session_ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:07.755394Z","iopub.execute_input":"2024-12-08T10:45:07.755750Z","iopub.status.idle":"2024-12-08T10:45:28.195463Z","shell.execute_reply.started":"2024-12-08T10:45:07.755712Z","shell.execute_reply":"2024-12-08T10:45:28.194632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Binary data","metadata":{}},{"cell_type":"markdown","source":"The distribution of binary values in each dataset","metadata":{}},{"cell_type":"code","source":"binary_columns = ['fullscreen', 'hq', 'music']\n# Summary statistics for binary columns in train_df and test_df\nfor col in binary_columns:\n    print(f'Summary for {col} in train_df:')\n    print(f'  - Proportion of 1s: {train_df[col].mean()}')\n    print(f'  - Proportion of 0s: {1 - train_df[col].mean()}')\n    print(f'  - Count of 1s: {train_df[col].sum()}')\n    print(f'  - Count of 0s: {len(train_df) - train_df[col].sum()}')\n    print('\\n')\n\n    print(f'Summary for {col} in test_df:')\n    print(f'  - Proportion of 1s: {test_df[col].mean()}')\n    print(f'  - Proportion of 0s: {1 - test_df[col].mean()}')\n    print(f'  - Count of 1s: {test_df[col].sum()}')\n    print(f'  - Count of 0s: {len(test_df) - test_df[col].sum()}')\n    print('\\n')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:28.196674Z","iopub.execute_input":"2024-12-08T10:45:28.197074Z","iopub.status.idle":"2024-12-08T10:45:28.503010Z","shell.execute_reply.started":"2024-12-08T10:45:28.197025Z","shell.execute_reply":"2024-12-08T10:45:28.502106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> **Observation:**\n> \n> - For binary features, the 'train' dataset have BOTH 2 VALUES. However, the 'test' dataset including just ONE of them.\n","metadata":{}},{"cell_type":"code","source":"# Plotting bar plots for each binary feature in train_df and test_df\nplt.figure(figsize=(12, 8))\nfor i, col in enumerate(binary_columns, 1):\n    plt.subplot(2, 3, i)\n    sns.countplot(data=train_df, x=col, palette='Set2')\n    plt.title(f'Distribution of {col} in train_df')\n    plt.xticks(ticks=[0, 1], labels=['0 (Off)', '1 (On)'])\n    plt.xlabel('')\n    plt.ylabel('Count')\n    \n    plt.subplot(2, 3, i+3)\n    sns.countplot(data=test_df, x=col, palette='Set2')\n    plt.title(f'Distribution of {col} in test_df')\n    plt.xticks(ticks=[0, 1], labels=['0 (Off)', '1 (On)'])\n    plt.xlabel('')\n    plt.ylabel('Count')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:28.504093Z","iopub.execute_input":"2024-12-08T10:45:28.504471Z","iopub.status.idle":"2024-12-08T10:45:33.487208Z","shell.execute_reply.started":"2024-12-08T10:45:28.504443Z","shell.execute_reply":"2024-12-08T10:45:33.486448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\n\n# Plot for train_df\nfor i, col in enumerate(binary_columns, 1):\n    plt.subplot(2, 3, i)\n    train_df[col].value_counts().plot.pie(autopct='%1.1f%%', colors=sns.color_palette('Set2', len(train_df[col].unique())), labels=['0 (Off)', '1 (On)'])\n    plt.title(f'Proportion of {col} in train_df')\n    plt.ylabel('')\n\n# Plot for test_df\nfor i, col in enumerate(binary_columns, 1):\n    plt.subplot(2, 3, i + 3)\n    test_df[col].value_counts().plot.pie(autopct='%1.1f%%', colors=sns.color_palette('Set2', len(test_df[col].unique())), labels=['0 (Off)', '1 (On)'])\n    plt.title(f'Proportion of {col} in test_df')\n    plt.ylabel('')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:33.488264Z","iopub.execute_input":"2024-12-08T10:45:33.488685Z","iopub.status.idle":"2024-12-08T10:45:34.945323Z","shell.execute_reply.started":"2024-12-08T10:45:33.488628Z","shell.execute_reply":"2024-12-08T10:45:34.944345Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Continous data","metadata":{}},{"cell_type":"code","source":"continuous_columns = ['session_id', 'index', 'elapsed_time', 'level', 'page', \n                      'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']\n# IQR method to detect outliers\nfor col in continuous_columns:\n    Q1 = train_df[col].quantile(0.25)\n    Q3 = train_df[col].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n\n    outliers_train = train_df[(train_df[col] < lower_bound) | (train_df[col] > upper_bound)]\n    outliers_test = test_df[(test_df[col] < lower_bound) | (test_df[col] > upper_bound)]\n\n    print(f'Outliers for {col} in train_df: {len(outliers_train)}')\n    print(f'Outliers for {col} in test_df: {len(outliers_test)}')\n    print('\\n')\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:34.946609Z","iopub.execute_input":"2024-12-08T10:45:34.946896Z","iopub.status.idle":"2024-12-08T10:45:42.057488Z","shell.execute_reply.started":"2024-12-08T10:45:34.946869Z","shell.execute_reply":"2024-12-08T10:45:42.056598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"level_group_df = train_df.groupby(\"level_group\")[\"page\"].value_counts().reset_index()\nlevel_group_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:42.058446Z","iopub.execute_input":"2024-12-08T10:45:42.058708Z","iopub.status.idle":"2024-12-08T10:45:43.759626Z","shell.execute_reply.started":"2024-12-08T10:45:42.058682Z","shell.execute_reply":"2024-12-08T10:45:43.758680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up the matplotlib figure\nplt.figure(figsize=(12, 8))\n\n# Create a bar plot\nsns.barplot(x='level_group', y='count', hue='page', data= level_group_df, palette='Set3')\n\n# Add title and labels\nplt.title('Distribution of Pages by Level Group', fontsize=18)\nplt.xlabel('Level Group', fontsize=14)\nplt.ylabel('Count', fontsize=14)\nplt.legend(title='Page')\n\n# Show the plot\nplt.show()\ndel level_group_df\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:43.760613Z","iopub.execute_input":"2024-12-08T10:45:43.760933Z","iopub.status.idle":"2024-12-08T10:45:44.246149Z","shell.execute_reply.started":"2024-12-08T10:45:43.760903Z","shell.execute_reply":"2024-12-08T10:45:44.245268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Labels Data","metadata":{}},{"cell_type":"markdown","source":"The `train_labels.csv` file includes two values:\n\n1. **session_id**: This does not equal the `session_id` from the training set. It is a combination of `<session_id>_<question #>`.\n2. **correct**: This is a flag indicating whether the answer is correct (1) or incorrect (0).\n\n**Note**: During gameplay, the player must eventually select the correct answer to continue. A \"correct answer\" here indicates that the player got the answer correct on their first attempt.","metadata":{}},{"cell_type":"code","source":"labels_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nprint(\"\\033[1mSample of labels data:\\033[0m\")\ndisplay(labels_df.head())\nprint()\nprint(\"\\033[1mStatistic description of labels data:\\033[0m\")\ndisplay(labels_df.describe().T.style.format(\"{:.2f}\").set_table_styles(\n    [{'selector': 'th', 'props': [('font-size', '12pt')]}]\n))\nprint()","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:45:44.250596Z","iopub.execute_input":"2024-12-08T10:45:44.250893Z","iopub.status.idle":"2024-12-08T10:45:44.581223Z","shell.execute_reply.started":"2024-12-08T10:45:44.250864Z","shell.execute_reply":"2024-12-08T10:45:44.580200Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> Mean of 0.71.\nThis suggests that approximately 71% of the questions were answered correctly, indicating a tendency towards success in the dataset, so the dataset is imbalanced. `question_number` distribution is equally among questions","metadata":{}},{"cell_type":"code","source":"# Split session_id into two columns\nlabels_df[['session_id_id', 'question_number']] = labels_df['session_id'].str.split('_', n=1, expand=True)\nlabels_df['session_id_id'] = labels_df['session_id_id'].astype(int)\n\n# Calculate mean correct percentage\nmean_correct = labels_df['correct'].mean() * 100\n\n# Calculate percentage of correct answers per question\nlabels_perc = (\n    labels_df.groupby('question_number')['correct']\n    .value_counts(normalize=True)\n    .mul(100)\n    .rename('Percent')\n    .reset_index()\n)\n\n# Filter for correct answers and sort by question number\nlabels_perc = labels_perc[labels_perc['correct'] == 1]\nlabels_perc['number'] = labels_perc['question_number'].str[1:].astype(int)\nlabels_perc.sort_values('number', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-12-08T10:45:44.582486Z","iopub.execute_input":"2024-12-08T10:45:44.582834Z","iopub.status.idle":"2024-12-08T10:45:45.522836Z","shell.execute_reply.started":"2024-12-08T10:45:44.582804Z","shell.execute_reply":"2024-12-08T10:45:45.521700Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the Seaborn bar plot\nplt.figure(figsize=(10, 10))\n\nsns.barplot(\n    x='number',\n    y='Percent',\n    data=labels_perc,\n    palette='YlGn',  # Set color palette\n)\n\n# Add a horizontal line for the mean correct percentage\nplt.axhline(\n    y=mean_correct, \n    color='coral', \n    linestyle='--', \n    label=f\"Average = {mean_correct:.1f}%\"\n)\n\n# Add text labels on the bars\nfor index, row in labels_perc.iterrows():\n    plt.text(\n        row['number'] - 1, \n        row['Percent'] + 1,  # Position text slightly above the bar\n        f'{row[\"Percent\"]:.1f}%', \n        ha='center', \n        va='bottom',\n        fontsize=10\n    )\n\n# Customize the plot\nplt.title(\"Share of Correct Answers by Question\", fontsize=18)\nplt.xlabel(\"Question Number\")\nplt.ylabel(\"Percent of Correct Answers\")\nplt.legend()\n\n# Display the plot\nplt.tight_layout()\nplt.show()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:45.524243Z","iopub.execute_input":"2024-12-08T10:45:45.524674Z","iopub.status.idle":"2024-12-08T10:45:46.014032Z","shell.execute_reply.started":"2024-12-08T10:45:45.524632Z","shell.execute_reply":"2024-12-08T10:45:46.013178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correct_answers_per_session = labels_df.groupby('session_id_id')['correct'].sum()\nmean_correct_answers_per_session = correct_answers_per_session.mean()\nmedian_correct_answers_per_session = correct_answers_per_session.median()\ncorrect_answers_per_session = correct_answers_per_session.value_counts()\nplt.figure(figsize=(20, 8))\ng = sns.barplot(x=correct_answers_per_session.index, y=correct_answers_per_session.values, color='rosybrown')\nplt.title('Distribution of games by number of correct answers', fontsize=18)\ng.set_xticklabels(['{}'.format(int(num + 1)) for num in g.get_xticks()])\ng.set(xlabel='Number of correct answers', ylabel='Count of sessions')\ng.axvline(x=mean_correct_answers_per_session-1, color=\"coral\")\ng.text(mean_correct_answers_per_session-1, 1500, f'Average ={round(mean_correct_answers_per_session, 1)}', rotation=90)\ng.axvline(x=median_correct_answers_per_session-1, color=\"peru\")\ng.text(median_correct_answers_per_session-1, 1500, f'Median ={round(median_correct_answers_per_session, 1)}', rotation=90)\nfor i, v in enumerate(correct_answers_per_session.sort_index().values):\n    plt.text(i-0.2, v, str(int(v)))\ndel correct_answers_per_session, mean_correct_answers_per_session, median_correct_answers_per_session, i, v\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:45:46.015287Z","iopub.execute_input":"2024-12-08T10:45:46.015692Z","iopub.status.idle":"2024-12-08T10:45:46.372416Z","shell.execute_reply.started":"2024-12-08T10:45:46.015654Z","shell.execute_reply":"2024-12-08T10:45:46.371548Z"}},"outputs":[],"execution_count":null}]}