{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing Libraries\nfrom scipy import stats\nimport pandas as pd\nimport numpy as np\nimport datetime\nimport warnings\nimport copy\nimport gc\nimport os\n\nimport seaborn as sns\nfrom cycler import cycler\nimport matplotlib as mpl\nimport matplotlib.dates as mdates\nfrom matplotlib import pyplot as plt, animation\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport statsmodels.api as sm\n\n\npd.set_option(\"display.max_rows\", 10)\npd.set_option(\"display.max_columns\", None)\n\nwarnings.simplefilter(\"ignore\")\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning) \ngc.enable()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:22:56.694136Z","iopub.execute_input":"2024-02-27T16:22:56.695110Z","iopub.status.idle":"2024-02-27T16:22:58.317869Z","shell.execute_reply.started":"2024-02-27T16:22:56.695073Z","shell.execute_reply":"2024-02-27T16:22:58.316869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Columns\n\n* session_id - the ID of the session the event took place in\n* index - the index of the event for the session\n* elapsed_time - how much time has passed (in milliseconds) between the start of the session and when the event was recorded\n* event_name - the name of the event type\n* name - the event name (e.g. identifies whether a notebook_click is is opening or closing the notebook)\n* level - what level of the game the event occurred in (0 to 22)\n* page - the page number of the event (only for notebook-related events)\n* room_coor_x - the coordinates of the click in reference to the in-game room (only for click events)\n* room_coor_y - the coordinates of the click in reference to the in-game room (only for click events)\n* screen_coor_x - the coordinates of the click in reference to the player’s screen (only for click events)\n* screen_coor_y - the coordinates of the click in reference to the player’s screen (only for click events)\n* hover_duration - how long (in milliseconds) the hover happened for (only for hover events)\n* text - the text the player sees during this event\n* fqid - the fully qualified ID of the event\n* room_fqid - the fully qualified ID of the room the event took place in\n* text_fqid - the fully qualified ID of the\n* fullscreen - whether the player is in fullscreen mode\n* hq - whether the game is in high-quality\n* music - whether the game music is on or off\n* level_group - which group of levels - and group of questions - this row belongs to (0-4, 5-12, 13-22)","metadata":{}},{"cell_type":"code","source":"directory = \"/kaggle/input/predict-student-performance-from-game-play\"\ntrain_path = \"/kaggle/input/predict-student-performance-from-game-play/train.csv\"\ntest_path = \"/kaggle/input/predict-student-performance-from-game-play/test.csv\"\ntrain_labels_path = \"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\"\nsample_submission_path = \"/kaggle/input/predict-student-performance-from-game-play/sample_submission.csv\"\n\ntrain = pd.read_csv(train_path)\ntest = pd.read_csv(test_path)\ntrain_labels = pd.read_csv(train_labels_path)\nsample_submission = pd.read_csv(sample_submission_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:38:37.472262Z","iopub.execute_input":"2024-02-27T16:38:37.472939Z","iopub.status.idle":"2024-02-27T16:39:53.529011Z","shell.execute_reply.started":"2024-02-27T16:38:37.472902Z","shell.execute_reply":"2024-02-27T16:39:53.528165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:25:05.023482Z","iopub.execute_input":"2024-02-27T16:25:05.023765Z","iopub.status.idle":"2024-02-27T16:25:05.047595Z","shell.execute_reply.started":"2024-02-27T16:25:05.023740Z","shell.execute_reply":"2024-02-27T16:25:05.046740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Example output\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:25:05.048917Z","iopub.execute_input":"2024-02-27T16:25:05.049294Z","iopub.status.idle":"2024-02-27T16:25:05.058123Z","shell.execute_reply.started":"2024-02-27T16:25:05.049258Z","shell.execute_reply":"2024-02-27T16:25:05.057174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Types","metadata":{}},{"cell_type":"code","source":"def get_dtypes(data_frame):\n    dtypes = data_frame.dtypes\n    dtypes = pd.DataFrame(dtypes).reset_index(drop=False)\n    dtypes = dtypes.rename(columns={\"index\": \"column\", 0: \"dtype\"})\n    \n    return dtypes\n\ndtypes = get_dtypes(train)\n\nwith pd.option_context(\"display.max_rows\", None):\n    display(dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:25:05.060148Z","iopub.execute_input":"2024-02-27T16:25:05.060456Z","iopub.status.idle":"2024-02-27T16:25:05.074667Z","shell.execute_reply.started":"2024-02-27T16:25:05.060433Z","shell.execute_reply":"2024-02-27T16:25:05.073773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimization:\n\nThere are a lot of techniques to optimize data pre-processing (e.g Parallelling and reducing data types)\n\n# First we will reduce the data type of string variables to 'category'. This will fascilitate faster loading of the data.","metadata":{}},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:25:05.075662Z","iopub.execute_input":"2024-02-27T16:25:05.075929Z","iopub.status.idle":"2024-02-27T16:26:32.966602Z","shell.execute_reply.started":"2024-02-27T16:25:05.075906Z","shell.execute_reply":"2024-02-27T16:26:32.965627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now we will try to reduce the memory usage","metadata":{}},{"cell_type":"code","source":"def get_memory_usage(data_frame, unit=\"MB\"):\n    memory_usage = data_frame.memory_usage().sum()\n    \n    if unit not in UNITS:\n        raise ValueError(f\"{unit} is not supported value for `unit`. Please, choose one of {UNITS}.\")\n        \n    memory_usage /= (1024 ** (UNITS.index(unit) + 1))\n    memory_usage = round(memory_usage, 2)\n    \n    return memory_usage\n\ndef reduce_memory_usage(data_frame, return_copy=True):\n    \"\"\"\n    References:\n        https://github.com/a-milenkin/ML_tricks_and_hacks/blob/main/dataframe_optimize.py\n    \"\"\"\n    \n    if return_copy:\n        data_frame = copy.deepcopy(data_frame)\n        \n    for column, dtype in zip(data_frame.columns, data_frame.dtypes):\n        dtype = str(dtype)\n        column_data = data_frame[column]\n        if any([_ in dtype for _ in (\"int\", \"float\")]):\n            column_min, column_max = column_data.min(), column_data.max()\n            if \"int\" in dtype:\n                if (column_min > np.iinfo(np.int8).min) and (column_max < np.iinfo(np.int8).max):\n                    column_data = column_data.astype(np.int8)\n                elif (column_min > np.iinfo(np.int16).min) and (column_max < np.iinfo(np.int16).max):\n                    column_data = column_data.astype(np.int16)\n                elif (column_min > np.iinfo(np.int32).min) and (column_max < np.iinfo(np.int32).max):\n                    column_data = column_data.astype(np.int32)\n                else:\n                    column_data = column_data.astype(np.int64)\n            else:\n                if (column_min > np.finfo(np.float16).min) and (column_max < np.finfo(np.float16).max):\n                    column_data = column_data.astype(np.float16)\n                elif (column_min > np.finfo(np.float32).min) and (column_max < np.finfo(np.float32).max):\n                    column_data = column_data.astype(np.float32)\n                else:\n                    column_data = column_data.astype(np.float64)\n        elif dtype == \"object\":\n            column_data = column_data.astype(\"category\")\n        elif \"datetime\" in dtype:\n            column_data = pd.to_datetime(column_data)\n            \n        data_frame[column] = column_data\n        \n    return data_frame\nUNITS = (\"KB\", \"MB\", \"GB\", \"TB\")\nunit = \"MB\"\n\nmemory_usage = get_memory_usage(train, unit=unit)\nprint(f\"Memory usage of train dataset is {memory_usage} {unit}.\")\n\ntrain = reduce_memory_usage(train, return_copy=False)\noptimized_memory_usage = get_memory_usage(train, unit=unit)\nprint(f\"Memory usage of optimized train dataset is {optimized_memory_usage} {unit}.\")\n\nmemory_usage_difference = (memory_usage - optimized_memory_usage)\nmemory_usage_percentage_difference = (memory_usage_difference / memory_usage) * 100\nmemory_usage_percentage_difference = round(memory_usage_percentage_difference, 2)\nprint(f\"Memory usage of train dataset was decreased by {memory_usage_percentage_difference}%!\")\n\nprint()\n\nmemory_usage = get_memory_usage(test, unit=unit)\nprint(f\"Memory usage of test dataset is {memory_usage} {unit}.\")\n\ntest = reduce_memory_usage(test, return_copy=False)\noptimized_memory_usage = get_memory_usage(test, unit=unit)\nprint(f\"Memory usage of optimized test dataset is {optimized_memory_usage} {unit}.\")\n\nmemory_usage_difference = (memory_usage - optimized_memory_usage)\nmemory_usage_percentage_difference = (memory_usage_difference / memory_usage) * 100\nmemory_usage_percentage_difference = round(memory_usage_percentage_difference, 2)\nprint(f\"Memory usage of test dataset was decreased by {memory_usage_percentage_difference}%!\")\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:36:06.163544Z","iopub.execute_input":"2024-02-27T16:36:06.163950Z","iopub.status.idle":"2024-02-27T16:36:30.925350Z","shell.execute_reply.started":"2024-02-27T16:36:06.163918Z","shell.execute_reply":"2024-02-27T16:36:30.924442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Duplicates","metadata":{}},{"cell_type":"code","source":"exclude_columns = (\"index\")\ncolumns = [column for column in train.columns if column not in exclude_columns]\nduplicates = train[train.duplicated(subset=columns)]\n\nnum_samples = len(train)\nnum_duplicates = len(duplicates)\nprint(f\"Number of duplicates: {num_duplicates}\")\n\nduplicates_percentage = (num_duplicates / num_samples) * 100\nduplicates_percentage = round(duplicates_percentage, 2)\nprint(f\"Percentage of duplicates: {duplicates_percentage}%\")\nprint()\n\nduplicates","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:28:39.046967Z","iopub.execute_input":"2024-02-27T16:28:39.047358Z","iopub.status.idle":"2024-02-27T16:29:56.301683Z","shell.execute_reply.started":"2024-02-27T16:28:39.047328Z","shell.execute_reply":"2024-02-27T16:29:56.300813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Missing values","metadata":{}},{"cell_type":"code","source":"def get_missing_values(data_frame, stat=\"count\"):\n    missing_values = data_frame.isnull().sum()\n    \n    if stat == \"percentage\":\n        num_samples = len(data_frame)\n        missing_values = (missing_values / num_samples) * 100\n    \n    columns = missing_values.index.tolist()\n    values = missing_values.values\n    \n    missing_values_data_frame = pd.DataFrame({\n        \"column\": columns,\n        \"missing_values\": values,\n    })\n    \n    return missing_values_data_frame\n\nmissing_values = get_missing_values(train, stat=\"percentage\")\n\nfigure = plt.figure(figsize=(20, 7))\naxis = figure.add_subplot()\naxis.grid(axis=\"y\", zorder=0)\nsns.barplot(x=\"column\", y=\"missing_values\", data=missing_values, label=\"Train dataset\", alpha=1.0, linewidth=2.0, ax=axis, zorder=2)\naxis.xaxis.set_tick_params(labelsize=14, rotation=40)\naxis.set_xlabel(\"column\", fontsize=17)\naxis.yaxis.set_tick_params(labelsize=12)\naxis.set_ylabel(\"missing values (%)\", fontsize=14)\naxis.set_yticks(range(0, 110, 10))\nfigure.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:36:43.270617Z","iopub.execute_input":"2024-02-27T16:36:43.271016Z","iopub.status.idle":"2024-02-27T16:36:44.530464Z","shell.execute_reply.started":"2024-02-27T16:36:43.270983Z","shell.execute_reply":"2024-02-27T16:36:44.529371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Removing unnecessary columns:\nSome columns don't have any information, so let's drop them","metadata":{}},{"cell_type":"code","source":"missed_columns = [\"fullscreen\", \"hq\", \"music\", \"page\", \"hover_duration\"]\ntrain = train.drop(missed_columns, axis=1)\ntest = test.drop(missed_columns, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:36:52.046978Z","iopub.execute_input":"2024-02-27T16:36:52.047680Z","iopub.status.idle":"2024-02-27T16:36:52.949226Z","shell.execute_reply.started":"2024-02-27T16:36:52.047647Z","shell.execute_reply":"2024-02-27T16:36:52.948266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of classes of correct and incorrect responses:\n\nAn imbalance among classes can cause the model to be biased to more balanced classes, i.e it will be overfitted and won’t generalize well for imbalance classes.","metadata":{}},{"cell_type":"code","source":"column = \"correct\"\n\nfigure = plt.figure(figsize=(10, 7))\naxis = figure.add_subplot()\naxis.grid(axis=\"y\", zorder=0)\nsns.countplot(x=column, data=train_labels, linewidth=2.0, ax=axis, zorder=2)\naxis.xaxis.set_tick_params(size=0, labelsize=14)\naxis.set_xlabel(column, fontsize=17)\naxis.yaxis.set_tick_params(size=0, labelsize=12)\naxis.set_ylabel(\"count\", fontsize=14)\naxis.tick_params(axis=\"both\", which=\"both\", length=0)\naxis.set_ylim(1)\nfigure.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:40:50.173315Z","iopub.execute_input":"2024-02-27T16:40:50.173714Z","iopub.status.idle":"2024-02-27T16:40:50.371092Z","shell.execute_reply.started":"2024-02-27T16:40:50.173682Z","shell.execute_reply":"2024-02-27T16:40:50.370170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see there is a slight class imbalance. To slightly smooth this problem: class-wise weights, oversampling or under-sampling, extracting external or generating synthetic data, and apply augmentations during training.\n\n","metadata":{}},{"cell_type":"markdown","source":"# Removing Outliers\n\n \"num_events\" and \"elapsed_time\" seem to have outliers","metadata":{}},{"cell_type":"code","source":"grouped_session = train.groupby(\"session_id\")\nsession_wise_statistics = pd.DataFrame({\n    \"num_events\": grouped_session.size(),\n    \"elapsed_time\": grouped_session[\"elapsed_time\"].mean(),\n})\nsession_wise_columns = session_wise_statistics.columns\nnum_columns = len(session_wise_columns)\n\ncolumns = 2\nrows = (num_columns // columns)  + 1\n\nfigure = plt.figure(figsize=(17, 10))\nfor index, column in enumerate(session_wise_columns):\n    data = session_wise_statistics[column].values\n    \n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"both\", zorder=0)\n    sns.kdeplot(x=data, fill=True, alpha=1.0,  linewidth=2, zorder=2, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.set_xlabel(column, fontsize=13, labelpad=7)\n    ylabel_text = \"density\"  if (index % columns) == 0 else \"\"\n    axis.set_ylabel(ylabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n\nfigure.tight_layout()\nfigure.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:42:18.487132Z","iopub.execute_input":"2024-02-27T16:42:18.487489Z","iopub.status.idle":"2024-02-27T16:42:20.066946Z","shell.execute_reply.started":"2024-02-27T16:42:18.487461Z","shell.execute_reply":"2024-02-27T16:42:20.065840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_outliers_mask = (session_wise_statistics[\"num_events\"] < 3000) & (session_wise_statistics[\"elapsed_time\"] < 0.02*1e8)\nnon_outliers_session_wise_statistics = session_wise_statistics[non_outliers_mask]\nsession_wise_columns = non_outliers_session_wise_statistics.columns\nnum_columns = len(session_wise_columns)\n\ncolumns = 2\nrows = (num_columns // columns)  + 1\n\nfigure = plt.figure(figsize=(17, 10))\nfor index, column in enumerate(session_wise_columns):\n    data = non_outliers_session_wise_statistics[column].values\n    \n    axis = figure.add_subplot(rows, columns, index+1)\n    axis.grid(axis=\"both\", zorder=0)\n    sns.kdeplot(x=data, fill=True, alpha=1.0, linewidth=2, zorder=2, ax=axis)\n    axis.xaxis.set_tick_params(labelsize=12)\n    axis.set_xlabel(column, fontsize=13, labelpad=7)\n    ylabel_text = \"density\"  if (index % columns) == 0 else \"\"\n    axis.set_ylabel(ylabel_text, fontsize=13)\n    axis.yaxis.set_tick_params(labelsize=10)\n\nfigure.suptitle(\"After removing outliers\", x=0.25, fontsize=25, fontweight=\"bold\")\nfigure.tight_layout()\nfigure.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:42:26.089137Z","iopub.execute_input":"2024-02-27T16:42:26.090001Z","iopub.status.idle":"2024-02-27T16:42:26.880268Z","shell.execute_reply.started":"2024-02-27T16:42:26.089964Z","shell.execute_reply":"2024-02-27T16:42:26.879327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Time taken by students as levels increase","metadata":{}},{"cell_type":"markdown","source":"We can observe that as the levels increase students take more time to complete them","metadata":{}},{"cell_type":"code","source":"level_vs_elapsed_time = pd.DataFrame(train.groupby(\"level\").agg({\"elapsed_time\": \"mean\"}))\nlevel_vs_elapsed_time = level_vs_elapsed_time.reset_index(drop=False)\nlevel_vs_elapsed_time.columns = [\"level\", \"elapsed_time\"]\nlevel_vs_elapsed_time = level_vs_elapsed_time.sort_values(by=\"level\", ascending=True)\n\nx, y = level_vs_elapsed_time[\"level\"].values, level_vs_elapsed_time[\"elapsed_time\"]\nmin_x, max_x = np.min(x), np.max(x)\n\nfigure = plt.figure(figsize=(20, 7))\naxis = figure.add_subplot()\naxis.grid(axis=\"both\", zorder=0)\nsns.lineplot(x=x, y=y, linewidth=2, alpha=1.0, ax=axis, zorder=2)\naxis.yaxis.set_tick_params(labelsize=14)\naxis.set_ylabel(\"elapsed_time\", fontsize=15)\naxis.set_xlabel(\"level\", fontsize=15)\naxis.xaxis.set_tick_params(labelsize=12)\naxis.yaxis.set_tick_params(labelsize=12)\naxis.fill_between(x, y,alpha=1.0, zorder=2)\naxis.set_xticks(range(min_x, max_x+1, 1))\naxis.set_xlim(min_x, max_x)\naxis.set_ylim(0.0)\nfigure.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-27T16:42:40.107269Z","iopub.execute_input":"2024-02-27T16:42:40.108018Z","iopub.status.idle":"2024-02-27T16:42:40.996593Z","shell.execute_reply.started":"2024-02-27T16:42:40.107984Z","shell.execute_reply":"2024-02-27T16:42:40.995523Z"},"trusted":true},"execution_count":null,"outputs":[]}]}