{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from scipy import stats\nimport pandas as pd\nimport numpy as np\nimport datetime\nimport warnings\nimport copy\nimport gc\nimport os\n\nimport seaborn as sns\nfrom cycler import cycler\nimport matplotlib as mpl\nimport matplotlib.dates as mdates\nfrom matplotlib import pyplot as plt, animation\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport statsmodels.api as sm\nfrom IPython.core.display import display, HTML\n\n\npd.set_option(\"display.max_rows\", 10)\npd.set_option(\"display.max_columns\", None)\n\nwarnings.simplefilter(\"ignore\")\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning) \ngc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:32.227648Z","iopub.execute_input":"2023-02-09T20:04:32.228067Z","iopub.status.idle":"2023-02-09T20:04:32.236179Z","shell.execute_reply.started":"2023-02-09T20:04:32.228028Z","shell.execute_reply":"2023-02-09T20:04:32.235159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# notebook's styles\ndef css_styling():\n    styles = \"\"\"\n    @import url('https://fonts.googleapis.com/css2?family=Inter:wght@400;500&display=swap');    \n    \n    /* Variables */\n    \n    :root {\n        --primary-color: #ef4444;\n        --text-color: #fff;\n        --text-font-size: 15px !important;\n        --primary-title-border-color: linear-gradient(to right, var(--primary-color) 0%, var(--primary-color) 50%, var(--background-color)) 0 1 100% 1;\n        --background-color: #202020;\n        --shadow-color: #6b7280;\n        --note-background-color: rgba(191,219,254, 0.1);\n        --note-text-color: #93c5fd;\n        --notebook-font: \"Inter\", sans-serif;\n        --quote-background-color: rgba(148,163,184,0.25);\n        --quote-text-color: #f3f4f6;\n        --warning-background-color: rgba(250, 204, 21, 0.1);\n        --warning-text-color: #fde047;\n        --formula-font-size: 25px;\n    }\n    \n    /* Notebook's defaults */\n    \n    * {\n        margin: 0;\n        padding: 0;\n        box-sizing: border-box;\n    }\n    \n    .jp-NotebookPanel-notebook, #notebook-container, .text_cell, .text_cell_render  {\n        background-color: var(--background-color) !important;\n        padding-bottom: 50px;\n    }\n    \n    #kaggle-portal-root-global + div {\n        display: none !important;\n    }\n    \n    .jp-OutputArea-output, .jp-OutputArea-output pre, .jp-OutputArea-output thead th {\n        color: var(--text-color) !important;\n    }\n    \n    /* Notebook's styles */\n    \n    /* Title's styles */\n    \n    .title_container {\n        background-color: var(--background-color); \n        position: relative;\n        border-width: 5px;\n        border-style: solid;\n        border-image: var(--primary-title-border-color);\n        margin-bottom: 10px;\n        font-family: var(--notebook-font);\n    }\n    \n    .border-0 {\n         border-width: 0px;\n         margin-bottom: 0px;\n    }\n    \n    .title span {\n        color: var(--text-color); \n        font-weight: 600;\n    }\n    \n    .paragraph_text {\n        color: var(--primary-color) !important;\n        font-family: var(--notebook-font);\n    }\n    \n    .paragraph {\n        color: var(--primary-color) !important;\n        font-family: var(--notebook-font);\n    }\n    \n    h1 .paragraph {\n        font-size: 35px;\n    }\n\n    h2 .paragraph {\n        font-size: 28px;\n    }\n    \n    /* Text's styles */\n    \n    .text {\n        color: var(--text-color) !important;\n        font-size: var(--text-font-size);\n        font-family: var(--notebook-font);\n    }\n    \n    .highlight_text {\n        background-color: var(--primary-color);\n        color: var(--text-color);\n        font-weight: bold;\n        padding: 1px 5px 1px 5px; \n        border-radius: 3px;\n        font-size: var(--text-font-size);\n    }\n    \n    .text_link {\n        color: #60a5fa !important;\n        font-size: var(--text-font-size);\n    }\n    \n    /* List's styles */\n    \n    .list ul li  {\n        color:  var(--text-color) !important;\n        font-size: var(--text-font-size);\n    }\n    \n    .list ul li a:not(.text_link) {\n        color: var(--text-color) !important;\n        font-size: var(--text-font-size);\n    }\n    \n    .list ul li {\n        margin-top: 0px !important;\n    }\n    \n    .list ul li::marker {\n        color: var(--primary-color);\n    }\n    \n    /* Image's styles */\n    \n    .image_container {\n        display: flex;\n        flex-direction: column;\n        align-items: center;\n        font-family: var(--notebook-font);\n    }\n     \n    .image_container img {\n        max-width: 99%;\n        box-shadow: 0px 0px 15px var(--shadow-color) !important;\n    }\n    \n    .image_container .text {\n        text-align: center;\n        margin-top: 10px;\n        color: #d1d5db !important;\n        font-size: var(--text-font-size);\n    }    \n\n    /* Attention blocks */\n\n    .attention {\n        padding: 10px;\n        border-radius: 5px;\n        font-family: var(--notebook-font);\n    }\n    \n    .note {\n        background-color: var(--note-background-color);\n    }\n    \n    .note_text {\n        color: var(--note-text-color);\n        font-size: var(--text-font-size);\n    }\n    \n    .warning {\n        background-color: var(--warning-background-color);\n    }\n    \n    .warning_text {\n        color: var(--warning-text-color);\n        font-size: var(--text-font-size);\n    }\n    \n    .quote {\n        background-color: var(--quote-background-color);\n    }\n    \n    .quote_text {\n        color: var(--quote-text-color);\n        font-size: var(--text-font-size);\n    }\n        \n    .formula {\n        font-size: var(--formula-font-size) !important;\n    }\n        \n    \"\"\"\n    return HTML(\"<style>\"+styles+\"</style>\")\n\ncss_styling()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:32.581961Z","iopub.execute_input":"2023-02-09T20:04:32.582285Z","iopub.status.idle":"2023-02-09T20:04:32.594626Z","shell.execute_reply.started":"2023-02-09T20:04:32.582256Z","shell.execute_reply":"2023-02-09T20:04:32.593692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# colors\nBACKGROUND_COLOR = \"#202020\"\nPRIMARY_COLOR = \"#ef4444\"\nTEXT_COLOR = \"#fff\"\nMALE_COLOR = \"#3b82f6\"\nFEMALE_COLOR = \"#d946ef\"\n\ncolors = [\"#ef4444\",  \"#f59e0b\",  \"#eab308\", \"#22c55e\", \"#60a5fa\", \"#4f46e5\", \"#9333ea\", \"#6b7280\"]\ncolors = sns.color_palette(colors)\n\npalette_colors = [\"#ef4444\", \"#7f1d1d\", \"#f5f5f1\", \"#ef4444\", \"#7f1d1d\"]\npalette = mpl.colors.LinearSegmentedColormap.from_list(\"\", palette_colors)\n\n# visualization styles\n\n# matplotlib\nvisualization_parameters = {\n    # figure styles\n    \"figure.facecolor\": BACKGROUND_COLOR,\n    \n    # axes styles\n    \"axes.facecolor\": BACKGROUND_COLOR,\n    \"axes.labelcolor\": TEXT_COLOR,\n    \"axes.prop_cycle\": cycler(color=colors),\n    \"axes.labelpad\": 7.0,\n    \"xtick.color\": TEXT_COLOR,\n    \"xtick.minor.size\": 0.0,\n    \"xtick.major.size\": 0.0,\n    \"xtick.major.pad\": 7.0,\n    \"xtick.minor.pad\": 7.0,\n    \"ytick.color\": TEXT_COLOR,\n    \"ytick.minor.size\": 0.0,\n    \"ytick.major.size\": 0.0,\n    \"ytick.major.pad\": 7.0,\n    \"ytick.minor.pad\": 7.0,\n    \n    # text styles\n    \"text.color\": TEXT_COLOR,\n    \"font.family\": \"serif\",\n    \n    \"axes.titlecolor\": TEXT_COLOR,\n    \"axes.titlelocation\": \"left\",\n    \"axes.titlepad\": 7.0,\n    \"axes.titleweight\": \"bold\",\n    \n    # spines styles\n    \"axes.spines.top\": False,\n    \"axes.spines.right\": False,\n    \"axes.spines.bottom\": False,\n    \"axes.spines.left\": False,\n    \n    # grid styles\n    \"grid.alpha\": 0.5,\n    \"grid.color\": TEXT_COLOR,\n    \"grid.linewidth\": 1.0,\n    \"grid.linestyle\": \"-\",\n    \n    # box plot styles\n    \"boxplot.boxprops.color\": PRIMARY_COLOR,\n    \"boxplot.capprops.color\": PRIMARY_COLOR,\n    \"boxplot.flierprops.color\": PRIMARY_COLOR,\n    \"boxplot.flierprops.markeredgecolor\": PRIMARY_COLOR,\n    \"boxplot.flierprops.markerfacecolor\": PRIMARY_COLOR,\n    \"boxplot.meanprops.color\": TEXT_COLOR,\n    \"boxplot.meanprops.markeredgecolor\": TEXT_COLOR,\n    \"boxplot.meanprops.markerfacecolor\": TEXT_COLOR,\n    \"boxplot.medianprops.color\": TEXT_COLOR,\n    \"boxplot.whiskerprops.color\": TEXT_COLOR,\n}\n\nboxplot_styles = {\n    \"capprops\": {\"color\": TEXT_COLOR, \"zorder\": 2},\n    \"boxprops\": {\"edgecolor\": TEXT_COLOR, \"zorder\": 2},\n    \"whiskerprops\": {\"color\": TEXT_COLOR, \"zorder\": 2},\n    \"flierprops\": {\"color\": TEXT_COLOR, \"marker\": \"o\", \"markerfacecolor\": PRIMARY_COLOR, \"markeredgecolor\": BACKGROUND_COLOR, \"zorder\": 2},\n    \"medianprops\": {\"color\": TEXT_COLOR, \"zorder\": 2},\n    \"meanprops\": {\"color\": TEXT_COLOR, \"zorder\": 2},\n}\n\nmpl.rcParams.update(visualization_parameters)\nmpl.rc(\"animation\", html=\"jshtml\")\n\nnumerical_distribution_legend_template = \"Mean: {mean:.2f}\\nMedian: {median:.2f}\\nSTD: {std:.2f}\\nSkew: {skew:.2f}\\nKurtosis: {kurtosis:.2f}\"\n\n# plotly\nlayout_styles = {\n    \"paper_bgcolor\": BACKGROUND_COLOR,\n    \"mapbox_style\": \"carto-darkmatter\",\n    \"margin\": {\"r\": 0, \"t\": 0, \"l\": 0, \"b\": 0},\n}\n\nremove_trace_text = \"<extra></extra>\"\n\nEPS = 1e-9","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:32.599609Z","iopub.execute_input":"2023-02-09T20:04:32.599902Z","iopub.status.idle":"2023-02-09T20:04:32.615209Z","shell.execute_reply.started":"2023-02-09T20:04:32.599876Z","shell.execute_reply":"2023-02-09T20:04:32.614302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# utilities\n\n# visualization utilities\ndef hide_spines(ax, spines=[\"top\", \"right\", \"left\", \"bottom\"]):\n    for spine in spines:\n        ax.spines[spine].set_visible(False)\n        \ndef pair_plot(columns, data, figsize=None, triangular_form=False, size_coef=3, **plot_args):\n    num_columns = len(columns)\n    if figsize is None:\n        width, height = num_columns * size_coef, num_columns * size_coef\n        figsize = (width, height)\n        \n    figure, axises = plt.subplots(num_columns, num_columns, figsize=figsize)\n    for i, x_column in enumerate(columns):\n        for j, y_column in enumerate(columns):\n            x_data = data[x_column].values\n            y_data = data[y_column].values\n            \n            axis = axises[i, j]\n            \n            if j > i and triangular_form:\n                axis.set_xticks([])\n                axis.set_yticks([])\n            else:\n                axis.grid(axis=\"both\", zorder=0)\n                sns.scatterplot(x=y_data, y=x_data, zorder=2, ax=axis, **plot_args)\n            \n                if i == (num_columns - 1):\n                    axis.set_xlabel(y_column, fontsize=13)\n                    \n                if j == 0:\n                    axis.set_ylabel(x_column, fontsize=13)\n                \n            hide_spines(axis)\n    \n    figure.tight_layout(w_pad=1, h_pad=1)\n    figure.show()\n\n    \ndef categorical_vs_categorical_plot(x, y, data, **plot_args):\n    data = data.groupby(x)[y].value_counts()\n    cat_vs_cat = np.array([list(index) for index in data.index])\n    counts = np.expand_dims(data.values, axis=-1)\n    data = np.concatenate([cat_vs_cat, counts], axis=-1)\n    data = pd.DataFrame(data, columns=[x, y, \"count\"])\n    data[\"count\"] = data[\"count\"].astype(int)\n    \n    max_count = data[\"count\"].max()\n    min_count = data[\"count\"].min()\n    \n    data = data.pivot(y, x, \"count\")\n    \n    plot_args[\"vmin\"] = plot_args.get(\"vmin\", min_count)\n    plot_args[\"vmax\"] = plot_args.get(\"vmax\", max_count)\n    plot_args[\"fmt\"] = plot_args.get(\"fmt\", \".0f\")\n    \n    sns.heatmap(data, **plot_args)\n    \ndef binary_plot(columns, data, **plot_args):\n    values = data[columns].mean(axis=0).values\n    x, y = columns, values\n    \n    orient = plot_args.get(\"orient\", \"v\")\n    if orient == \"h\":\n        x, y = values, columns\n        \n    sns.barplot(x=x, y=y, **plot_args)\n    \ndef seasonality_plot(column, datetime_column, data_frame, seasonality=\"month\", **plot_args):\n    data_frame = data_frame[[column, datetime_column]]\n    \n    if seasonality not in data_frame.columns:\n        data_frame[seasonality] = getattr(data_frame[datetime_column].dt, seasonality)\n        \n    sns.lineplot(x=seasonality, y=column, data=data_frame, **plot_args)\n    \ndef run_animation(images):    \n    \"\"\"\n    Runs matplotlib animation.\n    :param plt_ims: list of numpy.ndarray\n    :return: animation.FuncAnimation\n    \n    Source:\n        https://www.kaggle.com/code/sergiosaharovskiy/tps-oct-2022-viz-players-positions-animated\n    \"\"\"\n    \n    fig = plt.figure(figsize=(12,8))\n    \n    img = plt.imshow(images[0]) \n    plt.axis(\"off\")\n    plt.close()\n    \n    def frame(i):\n        img.set_array(images[i])\n        return [img]\n\n    return animation.FuncAnimation(fig, frame, frames=len(images), interval=70)\n    \n# data utilities\ndef get_missing_values(data_frame, stat=\"count\"):\n    missing_values = data_frame.isnull().sum()\n    \n    if stat == \"percentage\":\n        num_samples = len(data_frame)\n        missing_values = (missing_values / num_samples) * 100\n    \n    columns = missing_values.index.tolist()\n    values = missing_values.values\n    \n    missing_values_data_frame = pd.DataFrame({\n        \"column\": columns,\n        \"missing_values\": values,\n    })\n    \n    return missing_values_data_frame\n\ndef set_data_types(data_frame, numerical_columns=[], binary_columns=[], categorical_columns=[], datetime_columns=[], return_copy=False):\n    if return_copy:\n        data_frame = copy.deepcopy(data_frame)\n    \n    data_frame[numerical_columns] = data_frame[numerical_columns].astype(float)\n    data_frame[binary_columns] = data_frame[binary_columns].astype(bool)\n    data_frame[categorical_columns] = data_frame[categorical_columns].astype(str)\n    \n    if len(datetime_columns) > 0:\n        data_frame[datetime_columns] = pd.to_datetime(data_frame[datetime_columns])\n    \n    return data_frame\n\n\nUNITS = (\"KB\", \"MB\", \"GB\", \"TB\")\n\ndef get_memory_usage(data_frame, unit=\"MB\"):\n    memory_usage = data_frame.memory_usage().sum()\n    \n    if unit not in UNITS:\n        raise ValueError(f\"{unit} is not supported value for `unit`. Please, choose one of {UNITS}.\")\n        \n    memory_usage /= (1024 ** (UNITS.index(unit) + 1))\n    memory_usage = round(memory_usage, 2)\n    \n    return memory_usage\n\ndef reduce_memory_usage(data_frame, return_copy=True):\n    \"\"\"\n    References:\n        https://github.com/a-milenkin/ML_tricks_and_hacks/blob/main/dataframe_optimize.py\n    \"\"\"\n    \n    if return_copy:\n        data_frame = copy.deepcopy(data_frame)\n        \n    for column, dtype in zip(data_frame.columns, data_frame.dtypes):\n        dtype = str(dtype)\n        column_data = data_frame[column]\n        if any([_ in dtype for _ in (\"int\", \"float\")]):\n            column_min, column_max = column_data.min(), column_data.max()\n            if \"int\" in dtype:\n                if (column_min > np.iinfo(np.int8).min) and (column_max < np.iinfo(np.int8).max):\n                    column_data = column_data.astype(np.int8)\n                elif (column_min > np.iinfo(np.int16).min) and (column_max < np.iinfo(np.int16).max):\n                    column_data = column_data.astype(np.int16)\n                elif (column_min > np.iinfo(np.int32).min) and (column_max < np.iinfo(np.int32).max):\n                    column_data = column_data.astype(np.int32)\n                else:\n                    column_data = column_data.astype(np.int64)\n            else:\n                if (column_min > np.finfo(np.float16).min) and (column_max < np.finfo(np.float16).max):\n                    column_data = column_data.astype(np.float16)\n                elif (column_min > np.finfo(np.float32).min) and (column_max < np.finfo(np.float32).max):\n                    column_data = column_data.astype(np.float32)\n                else:\n                    column_data = column_data.astype(np.float64)\n        elif dtype == \"object\":\n            column_data = column_data.astype(\"category\")\n        elif \"datetime\" in dtype:\n            column_data = pd.to_datetime(column_data)\n            \n        data_frame[column] = column_data\n        \n    return data_frame\n\ndef get_dtypes(data_frame):\n    dtypes = data_frame.dtypes\n    dtypes = pd.DataFrame(dtypes).reset_index(drop=False)\n    dtypes = dtypes.rename(columns={\"index\": \"column\", 0: \"dtype\"})\n    \n    return dtypes\n\ndef get_cardinality(data_frame):\n    num_unique_values = data_frame.nunique()\n    num_unique_values = pd.DataFrame(num_unique_values).reset_index(drop=False)\n    num_unique_values = num_unique_values.rename(columns={\"index\": \"column\", 0: \"num_unique_values\"})\n    \n    return num_unique_values\n\n# the below code is copied from https://towardsdatascience.com/the-search-for-categorical-correlation-a1cf7f1888c9\ndef compute_cramers_v_function(x, y): \n    confusion_matrix = pd.crosstab(x,y)\n    chi2 = stats.chi2_contingency(confusion_matrix)[0]\n    n = confusion_matrix.sum().sum()\n    phi2 = chi2/n\n    r,k = confusion_matrix.shape\n    phi2corr = max(0, phi2-((k-1)*(r-1))/(n-1))\n    rcorr = r-((r-1)**2)/(n-1)\n    kcorr = k-((k-1)**2)/(n-1)\n    \n    return np.sqrt(phi2corr/min((kcorr-1),(rcorr-1)))\n\n\ndef get_cramers_v_correlation(data_frame, categorical_columns):\n    if categorical_columns is None:\n        categorical_columns_mask = [dtype in (\"object\", \"category\") for dtype in data_frame.dtypes]\n        categorical_columns = data_frame.columns[categorical_columns_mask]\n    \n    cat_data_frame = data_frame[categorical_columns]\n    \n    rows = []\n    for x in cat_data_frame:\n        col = []\n        for y in cat_data_frame :\n            try:\n                cramers = compute_cramers_v_function(cat_data_frame[x], cat_data_frame[y]) \n            except:\n                cramers = 0.0\n                \n            col.append(round(cramers,2))\n        rows.append(col)\n        \n    cramers_results = np.array(rows)\n    cramers_v_correlation = pd.DataFrame(cramers_results, columns=cat_data_frame.columns, index=cat_data_frame.columns)\n    \n    return cramers_v_correlation","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:32.679422Z","iopub.execute_input":"2023-02-09T20:04:32.680002Z","iopub.status.idle":"2023-02-09T20:04:32.716447Z","shell.execute_reply.started":"2023-02-09T20:04:32.679966Z","shell.execute_reply":"2023-02-09T20:04:32.715626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container\">\n    <h1 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>1.</span>\n        <span>Competition</span>\n    </h1>\n</div>\n<p class=\"text\">\n<i>Learning is meant to be fun, which is where game-based learning comes in. This educational approach allows students to engage with educational content inside a game framework, making it enjoyable and dynamic. Although game-based learning is being used in a growing number of educational settings, there are still a limited number of open datasets available to apply data science and learning analytic principles to improve game-based learning.<br><br>\nMost game-based learning platforms do not sufficiently make use of knowledge tracing to support individual students. Knowledge tracing methods have been developed and studied in the context of online learning environments and intelligent tutoring systems. But there has been less focus on knowledge tracing in educational games.<br><br>\nCompetition host Field Day Lab is a publicly-funded research lab at the Wisconsin Center for Educational Research. They design games for many subjects and age groups that bring contemporary research to the public, making use of the game data to understand how people learn. Field Day Lab's commitment to accessibility ensures all of its games are free and available to anyone. The lab also partners with ​​​nonprofits like The Learning Agency Lab, which is focused on developing science of learning-based tools and programs for the social good.<br><br>\nIf successful, you'll enable game developers to improve educational games and further support the educators who use these games with dashboards and analytic tools. In turn, we might see broader support for game-based learning platforms.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/overview/description\" class=\"text_link\">Context</a>\n</p><br>\n<div class=\"image_container\">\n    <img src=\"https://content.pbswisconsineducation.org/wp-content/uploads/2021/06/25221325/jw-facebook-playthegame.png\">\n    <span class=\"text\"></span>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>1.1.</span>\n        <span>Task</span>\n    </h2>\n</div>\n<p class=\"text\">\n<i>The goal of this competition is to predict student performance during game-based learning in real-time. You'll develop a model trained on one of the largest open datasets of game logs.<br><br>\nYour work will help advance research into knowledge-tracing methods for game-based learning. You'll be supporting developers of educational games to create more effective learning experiences for students.\n</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/overview\" class=\"text_link\">Goal of the Competition</a>\n</p>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>1.2.</span>\n        <span>Metric</span>\n    </h2>\n</div>\n<p class=\"text\">\n    <i>Submissions will be evaluated based on their F1 score.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/overview/evaluation\" class=\"text_link\">Evaluation</a><br>\n</p>\n<p class=\"text\">\n<i>\n    F-score or F-measure is a measure of a test's accuracy. It is calculated from the precision and recall of the test, where the precision is the number of true positive results divided by the number of all positive results, including those not identified correctly, and the recall is the number of true positive results divided by the number of all samples that should have been identified as positive. Precision is also known as positive predictive value, and recall is also known as sensitivity in diagnostic binary classification.<br><br>\nThe F1 score is the harmonic mean of the precision and recall. The more generic $ F_{\\beta } $ score applies additional weights, valuing one of precision or recall more than the other.<br><br>\nThe highest possible value of an F-score is 1.0, indicating perfect precision and recall, and the lowest possible value is 0, if either precision or recall are zero.\n</i> - <a href=\"https://en.wikipedia.org/wiki/F-score\" class=\"text_link\">Wikipedia</a><br>\n</p>\n<p class=\"text formula\">\n$ F_1 = \\frac{2}{recall^{-1} + precision^{-1}} = 2\\frac{precision \\cdot recall}{precision + recall} = \\frac{2tp}{2tp + fp + fn} $\n</p>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>1.3.</span>\n        <span>Efficiency Prize</span>\n    </h2>\n</div>\n<p class=\"text\">\n<i>We are hosting a second track that focuses on model efficiency, because highly accurate models are often computationally heavy. Such models have a stronger carbon footprint and frequently prove difficult to utilize in real-world educational contexts. We hope to use these models to help educational organizations, which have limited computational capabilities.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/overview/efficiency-prize-evaluation\" class=\"text_link\">Efficiency Prize</a>\n</p>\n<p class=\"text\">\n    <i>We compute a submission's efficiency score by:</i>\n</p>\n<p class=\"text formula\">\n$ \\text{Efficiency} = \\frac{1}{ \\text{Benchmark} - \\max\\text{F1} }\\text{F1} + \\frac{1}{32400}\\text{RuntimeSeconds} $\n</p>\n<p class=\"text\">\n    <i>where $ F1 $ is the submission's score on the main competition metric, $ Benchmark $ is the score of the benchmark sample_submission.csv, $ maxF1 $ is the maximum of all submissions on the Private Leaderboard, and $ RuntimeSeconds $ is the number of seconds it takes for the submission to be evaluated. The objective is to minimize the efficiency score.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/overview/efficiency-prize-evaluation\" class=\"text_link\">Evaluation Metric</a>\n</p>\n<div class=\"attention note\">\n    <span class=\"note_text\"><i>Note: this competition is aimed at producing models that are small and lightweight. We have introduced compute constraints to match - your VMs will have only 2 CPUs, 8GB of RAM, and no GPU available. You will still have a maximum of 9 hours to complete the task, but between the constraints and the efficiency prize there will be some interesting sub-problems to solve.</i></span>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container\">\n    <h1 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.</span>\n        <span>Data</span>\n    </h1>\n</div>\n<p class=\"text\">\n<i>This competition uses the Kaggle's time series API. Test data will be delivered in groupings that do not allow access to future data. The objective of this competition is to use time series data generated by an online educational game to determine whether players will answer questions correctly. There are three question checkpoints (level 4, level 12, and level 22), each with a number of questions. At each checkpoint, you will have access to all previous test data for that section.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/data\" class=\"text_link\">Dataset Description</a><br><br>\n<i>You have access to the training data and labels. There are 18 questions for each sessions - you are not given the answers, but are simply told whether the user for a particular session answered each question correctly.<br><br>\nWhen you are ready to predict, use the sample notebook to iterate over the test data, which is split as described above and served up as Pandas dataframes. Make your predictions for each group of questions - at the end of this process a submission.csv file will have been created for you. Simply submit your notebook.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/data\" class=\"text_link\">What files do I need?\n</a><br><br>\n<i>The training columns are as listed below. The label rows are identified with `session_id_question #`. Each session will have 18 rows, representing 18 questions.</i> - <a href=\"https://www.kaggle.com/competitions/predict-student-performance-from-game-play/data\" class=\"text_link\">What should I expect the data format to be?</a>\n</p>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.1.</span>\n        <span>Columns</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"list\">\n    <ul>\n        <li><b>session_id</b> - the ID of the session the event took place in</li>\n        <li><b>index</b> - the index of the event for the session</li>\n        <li><b>elapsed_time</b> - how much time has passed (in milliseconds) between the start of the session and when the event was recorded</li>\n        <li><b>event_name</b> - the name of the event type</li>\n        <li><b>name</b> - the event name (e.g. identifies whether a notebook_click is is opening or closing the notebook)</li>\n        <li><b>level</b> - what level of the game the event occurred in (0 to 22)</li>\n        <li><b>page</b> - the page number of the event (only for notebook-related events)</li>\n        <li><b>room_coor_x</b> - the coordinates of the click in reference to the in-game room (only for click events)</li>\n        <li><b>room_coor_y</b> - the coordinates of the click in reference to the in-game room (only for click events)</li>\n        <li><b>screen_coor_x</b> - the coordinates of the click in reference to the player’s screen (only for click events)</li>\n        <li><b>screen_coor_y</b> - the coordinates of the click in reference to the player’s screen (only for click events)</li>\n        <li><b>hover_duration</b> - how long (in milliseconds) the hover happened for (only for hover events)</li>\n        <li><b>text</b> - the text the player sees during this event</li>\n        <li><b>fqid</b> - the fully qualified ID of the event</li>\n        <li><b>room_fqid</b> - the fully qualified ID of the room the event took place in</li>\n        <li><b>text_fqid</b> - the fully qualified ID of the</li>\n        <li><b>fullscreen</b> - whether the player is in fullscreen mode</li>\n        <li><b>hq</b> - whether the game is in high-quality</li>\n        <li><b>music</b> - whether the game music is on or off</li>\n        <li><b>level_group</b> - which group of levels - and group of questions - this row belongs to (0-4, 5-12, 13-22)</li>\n    </ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.2.</span>\n        <span>Files</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"directory = \"/kaggle/input/predict-student-performance-from-game-play\"\ntrain_path = \"/kaggle/input/predict-student-performance-from-game-play/train.csv\"\ntrain_labels_path = \"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\"\ntest_path = \"/kaggle/input/predict-student-performance-from-game-play/test.csv\"\nsample_submission_path = \"/kaggle/input/predict-student-performance-from-game-play/sample_submission.csv\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:32.720272Z","iopub.execute_input":"2023-02-09T20:04:32.720614Z","iopub.status.idle":"2023-02-09T20:04:32.729484Z","shell.execute_reply.started":"2023-02-09T20:04:32.720589Z","shell.execute_reply":"2023-02-09T20:04:32.728546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(train_path)\ntrain_labels = pd.read_csv(train_labels_path)\ntest = pd.read_csv(test_path)\nsample_submission = pd.read_csv(sample_submission_path)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:32.732499Z","iopub.execute_input":"2023-02-09T20:04:32.732797Z","iopub.status.idle":"2023-02-09T20:04:58.896693Z","shell.execute_reply.started":"2023-02-09T20:04:32.732771Z","shell.execute_reply":"2023-02-09T20:04:58.895752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\n    <b>train.csv</b> - the training set\n</p>","metadata":{}},{"cell_type":"code","source":"train","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:58.898006Z","iopub.execute_input":"2023-02-09T20:04:58.898352Z","iopub.status.idle":"2023-02-09T20:04:58.927335Z","shell.execute_reply.started":"2023-02-09T20:04:58.898315Z","shell.execute_reply":"2023-02-09T20:04:58.926328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\n    <b>train_labels.csv</b> - correct value for all 18 questions for each session in the training set\n</p>","metadata":{}},{"cell_type":"code","source":"train_labels","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:58.930309Z","iopub.execute_input":"2023-02-09T20:04:58.931010Z","iopub.status.idle":"2023-02-09T20:04:58.944433Z","shell.execute_reply.started":"2023-02-09T20:04:58.930970Z","shell.execute_reply":"2023-02-09T20:04:58.943070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\n    <b>test.csv</b> - the test set\n</p>","metadata":{}},{"cell_type":"code","source":"test","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:58.946200Z","iopub.execute_input":"2023-02-09T20:04:58.946641Z","iopub.status.idle":"2023-02-09T20:04:58.976933Z","shell.execute_reply.started":"2023-02-09T20:04:58.946563Z","shell.execute_reply":"2023-02-09T20:04:58.975864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\n    <b>sample_submission.csv</b> - a sample submission file in the correct format\n</p>","metadata":{}},{"cell_type":"code","source":"sample_submission","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:58.978490Z","iopub.execute_input":"2023-02-09T20:04:58.978854Z","iopub.status.idle":"2023-02-09T20:04:58.996826Z","shell.execute_reply.started":"2023-02-09T20:04:58.978818Z","shell.execute_reply":"2023-02-09T20:04:58.995963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.3.</span>\n        <span>Data types</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"dtypes = get_dtypes(train)\n\nwith pd.option_context(\"display.max_rows\", None):\n    display(dtypes)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:58.997962Z","iopub.execute_input":"2023-02-09T20:04:58.998297Z","iopub.status.idle":"2023-02-09T20:04:59.013515Z","shell.execute_reply.started":"2023-02-09T20:04:58.998271Z","shell.execute_reply":"2023-02-09T20:04:59.011849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.4.</span>\n        <span>Cardinality</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"cardinality = get_cardinality(train)\n\nwith pd.option_context(\"display.max_rows\", None):\n    display(cardinality)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:04:59.015145Z","iopub.execute_input":"2023-02-09T20:04:59.015612Z","iopub.status.idle":"2023-02-09T20:05:09.646363Z","shell.execute_reply.started":"2023-02-09T20:04:59.015578Z","shell.execute_reply":"2023-02-09T20:05:09.645409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.5.</span>\n        <span>Duplicates</span>\n    </h2>\n</div>\n<p class=\"text\">\nSometimes some data are duplicated and because of duplicates, the model becomes slightly overfitted, because, roughly speaking, it looks at the same samples during training, i.e. don't observe other \"cases\".\n</p>","metadata":{}},{"cell_type":"code","source":"exclude_columns = (\"index\")\ncolumns = [column for column in train.columns if column not in exclude_columns]\nduplicates = train[train.duplicated(subset=columns)]\n\nnum_samples = len(train)\nnum_duplicates = len(duplicates)\nprint(f\"Number of duplicates: {num_duplicates}\")\n\nduplicates_percentage = (num_duplicates / num_samples) * 100\nduplicates_percentage = round(duplicates_percentage, 2)\nprint(f\"Percentage of duplicates: {duplicates_percentage}%\")\nprint()\n\nduplicates","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:09.647690Z","iopub.execute_input":"2023-02-09T20:05:09.648467Z","iopub.status.idle":"2023-02-09T20:05:42.776111Z","shell.execute_reply.started":"2023-02-09T20:05:09.648428Z","shell.execute_reply":"2023-02-09T20:05:42.775084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>3.6.</span>\n        <span>Missing values</span>\n    </h2>\n</div>\n<p class=\"text\">\n    <i>Missing data, or missing values, occur when no data value is stored for the variable in an observation. Missing data are a common occurrence and can have a significant effect on the conclusions that can be drawn from the data.</i> - <a href=\"https://en.wikipedia.org/wiki/Missing_data#:~:text=In%20statistics%2C%20missing%20data%2C%20or,be%20drawn%20from%20the%20data\" class=\"text_link\">Wikipedia</a>\n</p>\n<p class=\"text\">\nThere are many techniques to produce the imputation/filling of missing values. The basic technique is to use mean or median for missed numerical values and mode (the most frequent value) for missed categorical values respectively. More powerful and accurate approaches are k Nearest Neighbors (KNN) or even training additional model(s), where the target(s) is (are) the missed column(s).\n</p>\n<p class=\"text\">\nThe following papers provide more information about missing values and imputation techniques:\n</p>\n<div class=\"list\">\n    <ul>\n        <li><a href=\"https://scikit-learn.org/stable/modules/impute.html\">Imputation of missing values</a></li>\n        <li><a href=\"https://towardsdatascience.com/6-different-ways-to-compensate-for-missing-values-data-imputation-with-examples-6022d9ca0779\">6 Different Ways to Compensate for Missing Values In a Dataset (Data Imputation with examples)</a></li>\n    </ul>\n</div>\n<div class=\"attention note\">\n    <span class=\"note_text\">Tree-based models can handle missing values.</span>\n</div>","metadata":{}},{"cell_type":"code","source":"missing_values = get_missing_values(train, stat=\"percentage\")\n\nfigure = plt.figure(figsize=(20, 7))\naxis = figure.add_subplot()\naxis.grid(axis=\"y\", zorder=0)\nsns.barplot(x=\"column\", y=\"missing_values\", data=missing_values, color=colors[0], edgecolor=TEXT_COLOR, label=\"Train dataset\", alpha=1.0, linewidth=2.0, ax=axis, zorder=2)\naxis.xaxis.set_tick_params(labelsize=14, rotation=40)\naxis.set_xlabel(\"column\", fontsize=17)\naxis.yaxis.set_tick_params(labelsize=12)\naxis.set_ylabel(\"missing values (%)\", fontsize=14)\naxis.set_yticks(range(0, 110, 10))\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:58.159993Z","iopub.execute_input":"2023-02-09T20:05:58.160380Z","iopub.status.idle":"2023-02-09T20:06:01.752458Z","shell.execute_reply.started":"2023-02-09T20:05:58.160345Z","shell.execute_reply":"2023-02-09T20:06:01.751549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\nSome columns don't have any information, i.e the columns are fully missed, so let's drop them from the future analysis. \n</p>","metadata":{}},{"cell_type":"code","source":"missed_columns = [\"fullscreen\", \"hq\", \"music\", \"page\", \"hover_duration\"]\ntrain = train.drop(missed_columns, axis=1)\ntest = test.drop(missed_columns, axis=1)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.791475Z","iopub.status.idle":"2023-02-09T20:05:42.792291Z","shell.execute_reply.started":"2023-02-09T20:05:42.792026Z","shell.execute_reply":"2023-02-09T20:05:42.792055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container\">\n    <h1 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>4.</span>\n        <span>Optimization</span>\n    </h1>\n</div>\n<p class=\"text\">\n  Optimization as well as other steps of project development is the necessary intermediate step. Due to proper optimization productivity rate is increased, hence allows making experiments much faster, especially in Data Science competitions.<br><br>\nThere are a lot of techniques to optimize data pre-processing (e.g Parallelling and reducing data types) and training models (e.g. Intel® Extension). Also, operations can be applied on GPU machines with help of the RAPIDS library, which is made by NVIDIA. \n</p>","metadata":{}},{"cell_type":"code","source":"unit = \"MB\"\n\nmemory_usage = get_memory_usage(train, unit=unit)\nprint(f\"Memory usage of train dataset is {memory_usage} {unit}.\")\n\ntrain = reduce_memory_usage(train, return_copy=False)\noptimized_memory_usage = get_memory_usage(train, unit=unit)\nprint(f\"Memory usage of optimized train dataset is {optimized_memory_usage} {unit}.\")\n\nmemory_usage_difference = (memory_usage - optimized_memory_usage)\nmemory_usage_percentage_difference = (memory_usage_difference / memory_usage) * 100\nmemory_usage_percentage_difference = round(memory_usage_percentage_difference, 2)\nprint(f\"Memory usage of train dataset was decreased by {memory_usage_percentage_difference}%!\")\n\nprint()\n\nmemory_usage = get_memory_usage(test, unit=unit)\nprint(f\"Memory usage of test dataset is {memory_usage} {unit}.\")\n\ntest = reduce_memory_usage(test, return_copy=False)\noptimized_memory_usage = get_memory_usage(test, unit=unit)\nprint(f\"Memory usage of optimized test dataset is {optimized_memory_usage} {unit}.\")\n\nmemory_usage_difference = (memory_usage - optimized_memory_usage)\nmemory_usage_percentage_difference = (memory_usage_difference / memory_usage) * 100\nmemory_usage_percentage_difference = round(memory_usage_percentage_difference, 2)\nprint(f\"Memory usage of test dataset was decreased by {memory_usage_percentage_difference}%!\")\n\ngc.collect()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.793931Z","iopub.status.idle":"2023-02-09T20:05:42.794590Z","shell.execute_reply.started":"2023-02-09T20:05:42.794262Z","shell.execute_reply":"2023-02-09T20:05:42.794293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container\">\n    <h1 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.</span>\n        <span>Exploratory Data Analysis</span>\n    </h1>\n</div>\n<p class=\"text\">\nExploratory Data Analysis (EDA) is an integral and the most important stage during developing any Data Science projects. Exploratory Data Analysis can give important observations, relations and even ideas for further defining hypotheses and testing, which can be very beneficial for model development. From extracted observations, we can decide what metric, pre-processing, feature engineering, validation strategy, post-processing, model, etc, we should use, which potentially can improve model generalization, hence, improving of predictive ability. \n</p>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.1.</span>\n        <span>Target Variable Distribution</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"column = \"correct\"\n\nfigure = plt.figure(figsize=(10, 7))\naxis = figure.add_subplot()\naxis.grid(axis=\"y\", zorder=0)\nsns.countplot(x=column, data=train_labels, palette=colors,  edgecolor=TEXT_COLOR, linewidth=2.0, ax=axis, zorder=2)\naxis.xaxis.set_tick_params(size=0, labelsize=14)\naxis.set_xlabel(column, fontsize=17)\naxis.yaxis.set_tick_params(size=0, labelsize=12)\naxis.set_ylabel(\"count\", fontsize=14)\naxis.tick_params(axis=\"both\", which=\"both\", length=0)\naxis.set_ylim(1)\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.796579Z","iopub.status.idle":"2023-02-09T20:05:42.797270Z","shell.execute_reply.started":"2023-02-09T20:05:42.796998Z","shell.execute_reply":"2023-02-09T20:05:42.797022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">There is slight class imbalance. Because of imbalance model will be biased to more balanced classes, i.e will be overfitted and won’t generalize well for imbalance classes. There are tecniques to slightly smooth that problem: class-wise weights, oversampling or under-sampling, extracting external or generating synthetic data, and apply augmentations during training.</p>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.2.</span>\n        <span>Session-wise statistics</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"grouped_session = train.groupby(\"session_id\")\nsession_wise_statistics = pd.DataFrame({\n    \"num_events\": grouped_session.size(),\n    \"elapsed_time\": grouped_session[\"elapsed_time\"].mean(),\n})","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.798926Z","iopub.status.idle":"2023-02-09T20:05:42.799618Z","shell.execute_reply.started":"2023-02-09T20:05:42.799366Z","shell.execute_reply":"2023-02-09T20:05:42.799390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_wise_columns = session_wise_statistics.columns\nnum_columns = len(session_wise_columns)\n\ncolumns = 2\nrows = (num_columns // columns)  + 1\n\nfigure = plt.figure(figsize=(17, 10))\nfor index, column in enumerate(session_wise_columns):\n    data = session_wise_statistics[column].values\n    \n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"both\", zorder=0)\n    sns.kdeplot(x=data, fill=True, alpha=1.0, color=colors[0],  edgecolor=TEXT_COLOR,  linewidth=2, zorder=2, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.set_xlabel(column, fontsize=13, labelpad=7)\n    ylabel_text = \"density\"  if (index % columns) == 0 else \"\"\n    axis.set_ylabel(ylabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n\nfigure.tight_layout()\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.800952Z","iopub.status.idle":"2023-02-09T20:05:42.801634Z","shell.execute_reply.started":"2023-02-09T20:05:42.801391Z","shell.execute_reply":"2023-02-09T20:05:42.801415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_outliers_mask = (session_wise_statistics[\"num_events\"] < 3000) & (session_wise_statistics[\"elapsed_time\"] < 0.02*1e8)\nnon_outliers_session_wise_statistics = session_wise_statistics[non_outliers_mask]\nsession_wise_columns = non_outliers_session_wise_statistics.columns\nnum_columns = len(session_wise_columns)\n\ncolumns = 2\nrows = (num_columns // columns)  + 1\n\nfigure = plt.figure(figsize=(17, 10))\nfor index, column in enumerate(session_wise_columns):\n    data = non_outliers_session_wise_statistics[column].values\n    \n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"both\", zorder=0)\n    sns.kdeplot(x=data, fill=True, alpha=1.0, color=colors[0],  edgecolor=TEXT_COLOR,  linewidth=2, zorder=2, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.set_xlabel(column, fontsize=13, labelpad=7)\n    ylabel_text = \"density\"  if (index % columns) == 0 else \"\"\n    axis.set_ylabel(ylabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n\nfigure.suptitle(\"Session-wise statistics without outliers\", x=0.25, fontsize=25, fontweight=\"bold\")\nfigure.tight_layout()\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.802933Z","iopub.status.idle":"2023-02-09T20:05:42.803625Z","shell.execute_reply.started":"2023-02-09T20:05:42.803371Z","shell.execute_reply":"2023-02-09T20:05:42.803395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.3.</span>\n        <span>Numerical columns distributions</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"common_numerical_columns = [\"room_coor_x\", \"room_coor_y\", \"screen_coor_x\", \"screen_coor_y\"]\nnum_common_numerical_columns_columns = len(common_numerical_columns)\n\ncolumns = 2\nrows = (num_common_numerical_columns_columns // columns)  + 1\n\nfigure = plt.figure(figsize=(17, 10))\nfor index, column in enumerate(common_numerical_columns):\n    data = train[column].values\n    \n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"both\", zorder=0)\n    sns.kdeplot(x=data, fill=True, alpha=1.0, color=colors[0],  edgecolor=TEXT_COLOR,  linewidth=2.0, zorder=2, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.set_xlabel(column, fontsize=13, labelpad=7)\n    ylabel_text = \"density\"  if (index % columns) == 0 else \"\"\n    axis.set_ylabel(ylabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n\nfigure.tight_layout(w_pad=2, h_pad=2)\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.804938Z","iopub.status.idle":"2023-02-09T20:05:42.805630Z","shell.execute_reply.started":"2023-02-09T20:05:42.805375Z","shell.execute_reply":"2023-02-09T20:05:42.805400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.4.</span>\n        <span>Categorical columns distributions</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"common_categorical_columns = [\"event_name\", \"name\", \"level\", \"level_group\", \"room_fqid\"]\nnum_common_categorical_columns = len(common_categorical_columns)\n\ncolumns = 2\nrows = (num_common_categorical_columns // columns)  + 1\nrotate_ticks_threshold = 5\n\nfigure = plt.figure(figsize=(15, 17))\nfigure_colors = []\nfor index, column in enumerate(common_categorical_columns):\n    data = train[column].values\n    \n    axis_colors = colors\n    if index < len(figure_colors):\n        axis_colors = figure_colors[index]\n    \n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"y\", zorder=0)\n    sns.countplot(x=data, fill=True, palette=axis_colors, alpha=1.0, edgecolor=TEXT_COLOR, linewidth=2.0, zorder=2, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.set_xlabel(column, fontsize=13)\n    ylabel_text = \"count\"  if (index % columns) == 0 else \"\"\n    axis.set_ylabel(ylabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n    \n    num_categories = len(set(data))\n    if num_categories > rotate_ticks_threshold:\n        axis.tick_params(axis=\"x\", labelrotation=90)\n    \n    axis.set_ylim(1)\n    \nfigure.tight_layout(h_pad=1, w_pad=2)\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.806943Z","iopub.status.idle":"2023-02-09T20:05:42.807635Z","shell.execute_reply.started":"2023-02-09T20:05:42.807384Z","shell.execute_reply":"2023-02-09T20:05:42.807408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.5.</span>\n        <span>Level vs elapsed time</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"level_vs_elapsed_time = pd.DataFrame(train.groupby(\"level\").agg({\"elapsed_time\": \"mean\"}))\nlevel_vs_elapsed_time = level_vs_elapsed_time.reset_index(drop=False)\nlevel_vs_elapsed_time.columns = [\"level\", \"elapsed_time\"]\nlevel_vs_elapsed_time = level_vs_elapsed_time.sort_values(by=\"level\", ascending=True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.808936Z","iopub.status.idle":"2023-02-09T20:05:42.809630Z","shell.execute_reply.started":"2023-02-09T20:05:42.809380Z","shell.execute_reply":"2023-02-09T20:05:42.809405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, y = level_vs_elapsed_time[\"level\"].values, level_vs_elapsed_time[\"elapsed_time\"]\nmin_x, max_x = np.min(x), np.max(x)\n\nfigure = plt.figure(figsize=(20, 7))\naxis = figure.add_subplot()\naxis.grid(axis=\"both\", zorder=0)\nsns.lineplot(x=x, y=y, color=TEXT_COLOR, linewidth=2, alpha=1.0, ax=axis, zorder=2)\naxis.yaxis.set_tick_params(labelsize=14)\naxis.set_ylabel(\"elapsed_time\", fontsize=15)\naxis.set_xlabel(\"level\", fontsize=15)\naxis.xaxis.set_tick_params(labelsize=12)\naxis.yaxis.set_tick_params(labelsize=12)\naxis.fill_between(x, y, color=PRIMARY_COLOR, edgecolor=TEXT_COLOR, alpha=1.0, zorder=2)\naxis.set_xticks(range(min_x, max_x+1, 1))\naxis.set_xlim(min_x, max_x)\naxis.set_ylim(0.0)\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.810934Z","iopub.status.idle":"2023-02-09T20:05:42.811623Z","shell.execute_reply.started":"2023-02-09T20:05:42.811374Z","shell.execute_reply":"2023-02-09T20:05:42.811398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\nWe are observing trend/correlation: more level => more time required to complete. \n</p>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.6.</span>\n        <span>Text length (number of words) distribution</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"texts = train[~train[\"text\"].isna()][\"text\"].values\nnum_words = np.array([len(str(text).split()) for text in texts])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.812910Z","iopub.status.idle":"2023-02-09T20:05:42.813636Z","shell.execute_reply.started":"2023-02-09T20:05:42.813348Z","shell.execute_reply":"2023-02-09T20:05:42.813372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figure = plt.figure(figsize=(15, 5))\naxis = figure.add_subplot()\naxis.grid(axis=\"both\", zorder=0)\nsns.countplot(x=num_words, fill=True, alpha=1.0, color=colors[0],  edgecolor=TEXT_COLOR,  linewidth=2.0, zorder=2, ax=axis)\naxis.xaxis.set_tick_params(labelsize=12)\naxis.set_xlabel(\"num_words\", fontsize=13, labelpad=7)\naxis.set_ylabel(\"count\", fontsize=13)\naxis.yaxis.set_tick_params(labelsize=10)\naxis.set_ylim(EPS)\n    \nfigure.tight_layout()\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.815252Z","iopub.status.idle":"2023-02-09T20:05:42.815988Z","shell.execute_reply.started":"2023-02-09T20:05:42.815742Z","shell.execute_reply":"2023-02-09T20:05:42.815766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.7.</span>\n        <span>Level intersections</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"level_intersect_columns = [\"room_fqid\", \"event_name\", \"name\"]\nnum_level_intersect_columns = len(level_intersect_columns)\n\ncolumns = 1\nrows = (num_level_intersect_columns // columns) + 1\nfigure = plt.figure(figsize=(17, 25))\nfor index, column in enumerate(level_intersect_columns):\n    axis = figure.add_subplot(rows, columns, index+1)\n    categorical_vs_categorical_plot(x=\"level\", y=column, data=train, annot=True, linecolor=TEXT_COLOR, linewidth=2.0, cbar=False, cmap=palette, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    xlabel_text = \"level\" if index >= (num_level_intersect_columns - columns) else \"\" \n    axis.set_xlabel(xlabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n    axis.set_ylabel(column, fontsize=13)\n    \nfigure.tight_layout(w_pad=2, h_pad=2)\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.817308Z","iopub.status.idle":"2023-02-09T20:05:42.818085Z","shell.execute_reply.started":"2023-02-09T20:05:42.817811Z","shell.execute_reply":"2023-02-09T20:05:42.817838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>5.8.</span>\n        <span>Session</span>\n    </h2>\n</div>\n<p class=\"text\">\nLet's take a view on the one game session. \n</p>","metadata":{}},{"cell_type":"code","source":"def visualize_session(session, plot_directory=None, plot_filename_format=\"{session_id}_{index}.png\"):\n    if plot_directory is None:\n        plot_directory = str(session[\"session_id\"].values[0])\n        if not os.path.exists(plot_directory):\n            os.mkdir(plot_directory)\n    \n    # screen\n    screen_x_min, screen_y_min = session[\"screen_coor_x\"].min(), session[\"screen_coor_y\"].min()\n    screen_x_max, screen_y_max = session[\"screen_coor_x\"].max(), session[\"screen_coor_y\"].max()\n    screen_width, screen_height = screen_x_max - screen_x_min, screen_y_max - screen_y_min\n    screen_padding = 25\n    \n    # room\n    room_x_min, room_y_min = session[\"room_coor_x\"].min(), session[\"room_coor_y\"].min()\n    room_x_max, room_y_max = session[\"room_coor_x\"].max(), session[\"room_coor_y\"].max()\n    room_width, room_height = room_x_max - room_x_min, room_y_max - room_y_min\n    \n    plot_pathes = []\n    for index, sample in session.iterrows():\n        figure = plt.figure(figsize=(12, 10))\n        axis = figure.add_subplot()\n        \n        # environment\n        screen_rectangle = mpl.patches.Rectangle(xy=(screen_x_min, screen_y_min), width=screen_width, height=screen_height, edgecolor=\"#fff\", label=\"Screen\", color=colors[-1])\n        axis.add_patch(screen_rectangle)\n        \n        room_rectangle = mpl.patches.Rectangle(xy=(room_x_min, room_y_min), width=room_width, height=room_height, edgecolor=\"#fff\", label=\"Room\", color=colors[2])\n        axis.add_patch(room_rectangle)\n        \n        axis.set_xlim(screen_x_min - screen_padding, screen_x_max + screen_padding)\n        axis.set_ylim(screen_y_min - screen_padding, screen_y_max + screen_padding)\n        axis.set(xticks=[], yticks=[])\n        axis.legend()\n        \n        # events\n        room_x, room_y = sample[\"room_coor_x\"], sample[\"room_coor_y\"]\n        screeen_x, screeen_y = sample[\"screen_coor_x\"], sample[\"screen_coor_y\"]\n        axis.plot(room_x, room_y, color=colors[0], marker=\"o\", markersize=5)\n        axis.plot(screeen_x, screeen_y, color=colors[3], marker=\"o\", markersize=5)\n        axis.set_title(\"FQID: {fqid}, Level: {level}, Index: {index}, Event name: {event_name}\".format(**sample.to_dict()), fontsize=20)\n        \n        # saving\n        plot_filename = plot_filename_format.format(**sample.to_dict())\n        plot_path = os.path.join(plot_directory, plot_filename)\n        figure.savefig(plot_path)\n        plt.close(figure) \n        \n        plot_pathes.append(plot_path)\n    \n    plt_ims = [plt.imread(plot_path) for plot_path in plot_pathes]\n    return run_animation(plt_ims)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.819394Z","iopub.status.idle":"2023-02-09T20:05:42.820108Z","shell.execute_reply.started":"2023-02-09T20:05:42.819862Z","shell.execute_reply":"2023-02-09T20:05:42.819886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_id = 20110112402567744\nsession = train[train[\"session_id\"] == session_id].iloc[:50]\n\nvisualize_session(session)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.821363Z","iopub.status.idle":"2023-02-09T20:05:42.822084Z","shell.execute_reply.started":"2023-02-09T20:05:42.821832Z","shell.execute_reply":"2023-02-09T20:05:42.821858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"title_container\">\n    <h1 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>6.</span>\n        <span>Feature Engineering</span>\n    </h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"title_container border-0\">\n    <h2 class=\"title\">\n        <span class=\"paragraph_text\"><span class=\"paragraph\">§</span>6.1.</span>\n        <span>session_id Reverse Engineering</span>\n    </h2>\n</div>","metadata":{}},{"cell_type":"code","source":"def extract_date_from_session_id(data_frame, return_copy=False):\n    \"\"\"\n    https://www.kaggle.com/code/pdnartreb/session-id-reverse-engineering\n    \"\"\"\n    \n    if return_copy:\n        data_frame = copy.deepcopy(data_frame)\n    \n    data_frame[\"year\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[:2]))\n    data_frame[\"month\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[2:4]) + 1)\n    data_frame[\"weekday\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[4:6]))\n    data_frame[\"hour\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[6:8]))\n    data_frame[\"minute\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[8:10]))\n    data_frame[\"second\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[10:12]))\n    data_frame[\"ms\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[12:15]))\n    data_frame[\"?\"] = data_frame[\"session_id\"].apply(lambda x: int(str(x)[15:17]))\n    \n    return data_frame","metadata":{"execution":{"iopub.status.busy":"2023-02-09T20:05:42.823355Z","iopub.status.idle":"2023-02-09T20:05:42.824066Z","shell.execute_reply.started":"2023-02-09T20:05:42.823814Z","shell.execute_reply":"2023-02-09T20:05:42.823838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = extract_date_from_session_id(train)","metadata":{"execution":{"iopub.status.busy":"2023-02-09T20:05:42.825319Z","iopub.status.idle":"2023-02-09T20:05:42.826033Z","shell.execute_reply.started":"2023-02-09T20:05:42.825779Z","shell.execute_reply":"2023-02-09T20:05:42.825803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datetime_columns = [\"year\", \"month\", \"weekday\", \"hour\"]\nnum_datetime_columns = len(datetime_columns)\n\ncolumns = 2\nrows = int(num_datetime_columns // columns) + 1\n\nfigure = plt.figure(figsize=(15, 12))\nfor index, column in enumerate(datetime_columns):\n    column_grouped = train.groupby(column).agg({\"elapsed_time\": \"mean\"})\n    column_grouped = column_grouped.reset_index(drop=False)\n    column_grouped.columns = [column, \"elapsed_time\"]\n    \n    x, y = column_grouped[column].values, column_grouped[\"elapsed_time\"]\n    min_x, max_x = np.min(x), np.max(x)\n\n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"both\", zorder=0)\n    sns.lineplot(x=x, y=y, color=TEXT_COLOR, linewidth=2.0, alpha=1.0, ax=axis, zorder=2)\n    axis.yaxis.set_tick_params(labelsize=14)\n    axis.set_ylabel(\"elapsed_time\", fontsize=15)\n    axis.set_xlabel(column, fontsize=15)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.yaxis.set_tick_params(labelsize=12)\n    axis.fill_between(x, y, color=PRIMARY_COLOR, edgecolor=TEXT_COLOR, alpha=1.0, zorder=2)\n    axis.set_xticks(range(min_x, max_x+1, 1))\n    axis.set_xlim(min_x, max_x)\n    axis.set_ylim(0.0)\n\nfigure.tight_layout()\nfigure.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-02-09T20:05:42.827394Z","iopub.status.idle":"2023-02-09T20:05:42.828153Z","shell.execute_reply.started":"2023-02-09T20:05:42.827895Z","shell.execute_reply":"2023-02-09T20:05:42.827921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p class=\"text\">\n  The game is mostly played at the beginning of the school year (August, September, October, November, and December months) and during the school day (from 8 a.m. to 15p.m., then increasingly are appeared in the evening (maybe children do homework). Each year, the mean gaming time is increasing linearly.\n</p>","metadata":{}}]}