{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os \nfor dirname, _, filenames in os.walk('/kaggle/input/predict-student-performance-from-game-play'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-04T00:37:53.971367Z","iopub.execute_input":"2023-05-04T00:37:53.971853Z","iopub.status.idle":"2023-05-04T00:37:53.986937Z","shell.execute_reply.started":"2023-05-04T00:37:53.971819Z","shell.execute_reply":"2023-05-04T00:37:53.985902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-05-04T00:37:53.988781Z","iopub.execute_input":"2023-05-04T00:37:53.989549Z","iopub.status.idle":"2023-05-04T00:37:54.001534Z","shell.execute_reply.started":"2023-05-04T00:37:53.989511Z","shell.execute_reply":"2023-05-04T00:37:53.999156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***DEFINING TYPES AND USED COLUMNS***","metadata":{}},{"cell_type":"code","source":"PATH = '/kaggle/input/predict-student-performance-from-game-play'\n\ndtypes = {\"session_id\": 'int64',\n          \"index\": np.int16,\n          \"elapsed_time\": np.int32,\n          \"event_name\": 'category',\n          \"name\": 'category',\n          \"level\": np.int8,\n          \"page\": np.float16,\n          \"room_coor_x\": np.float16,\n          \"room_coor_y\": np.float16,\n          \"screen_coor_x\": np.float16,\n          \"screen_coor_y\": np.float16,\n          \"hover_duration\": np.float32,\n          \"text\": 'category',\n          \"fqid\": 'category',\n          \"room_fqid\": 'category',\n          \"text_fqid\": 'category',\n          \"fullscreen\": np.int8,\n          \"hq\": np.int8,\n          \"music\": np.int8,\n          \"level_group\": 'category'\n          }\n# Specify the list of columns you are using\nuse_col = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page', 'room_coor_x', 'room_coor_y', \n           'screen_coor_x', 'screen_coor_y', 'hover_duration', 'text', 'fqid', 'room_fqid', 'text_fqid', 'fullscreen', 'hq', 'music', 'level_group']\n","metadata":{"execution":{"iopub.status.busy":"2023-05-04T00:37:54.003786Z","iopub.execute_input":"2023-05-04T00:37:54.004512Z","iopub.status.idle":"2023-05-04T00:37:54.019811Z","shell.execute_reply.started":"2023-05-04T00:37:54.004462Z","shell.execute_reply":"2023-05-04T00:37:54.018616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_original = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes, usecols=use_col)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T00:37:54.021402Z","iopub.execute_input":"2023-05-04T00:37:54.022094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_original = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\ntrain_labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nsample_submission = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/sample_submission.csvv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ***Feature Engineering*** ","metadata":{}},{"cell_type":"code","source":"my_cols = ['index', 'session_id', 'elapsed_time', 'event_name', 'name', 'room_coor_x', 'room_coor_y', 'fullscreen', 'text', 'text_fqid', 'hq', 'page', 'level']\nmy_train_df = train_original[my_cols]\nmy_test_df = test_original[my_cols]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sessions = train_original['session_id'].unique().tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fixing Elapsed Time","metadata":{}},{"cell_type":"code","source":"elapsed_list = train_original['elapsed_time'].values.tolist()\n\nfor i in range(1, len(elapsed_list)):\n  if elapsed_list[i] < elapsed_list[i-1] and elapsed_list[i+1] != 0:\n    for ii in range(i,len(elapsed_list)):\n      if elapsed_list[ii] == 0: break\n      if elapsed_list[ii] >= elapsed_list[i-1]:\n        elapsed_list[i] = int(round((elapsed_list[ii] + elapsed_list[i-1])/2))\n        break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_original['elapsed_time2'] = elapsed_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting Time and Date of start of dataset per session","metadata":{}},{"cell_type":"code","source":"year, month, weekday, hour, minute, second, ms = [], [], [], [], [], [], []\n\nfor c in train_original['session_id'].unique().tolist():\n  year.append(int(str(c)[:2]))\n  month.append(int(str(c)[2:4]))\n  weekday.append(int(str(c)[4:6]))\n  hour.append(int(str(c)[6:8]))\n  minute.append(int(str(c)[8:10]))\n  second.append(int(str(c)[10:12]))\n  ms.append(int(str(c)[12:15]))\n\ndate_df = pd.DataFrame(train_original['session_id'].unique().tolist(), columns =['session_id'])\ndate_df['year'], date_df['month'], date_df['weekday'], date_df['hour'], date_df['minute'], date_df['second'], date_df['ms'] = year, month, weekday, hour, minute, second, ms\ndate_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_date_info(df):\n  sessions_initial_elapsed_time = df.groupby('session_id')['elapsed_time2'].agg('min').tolist()\n\n  year = date_df['year'].tolist()\n  month = date_df['month'].tolist()\n  day = date_df['weekday'].tolist()\n  hour = date_df['hour'].tolist()\n  minute = date_df['minute'].tolist()\n\n  session_weekday = []\n  session_hour = []\n  session_month = []\n  session_year = []\n\n  for i in range(len(sessions_initial_elapsed_time)): #shape : 23562, elements: number of events of each session\n    ms = sessions_initial_elapsed_time[i]\n\n    hours = ms/3.6e+6\n    days = hours//24\n    hours = hours - days*24\n    hours = round(hours + hour[i])\n\n    weekday = day[i] + days - (((day[i] + days)//6)*6)\n    session_weekday.append(weekday)\n    a = 0 #chech if minute is over 30; if so, round to one hour above\n    if minute[i] > 30: a=1\n    session_hour.append(hours+a)\n    session_month.append(month[i])\n    session_year.append(year[i])\n\n  s1 = pd.Series(session_weekday, index = sessions)\n  s1.name = 'session_weekday'\n  s1 = s1.rename_axis(\"session_id\")\n  s2 = pd.Series(session_hour, index = sessions)\n  s2.name = 'session_hour'\n  s2 = s2.rename_axis(\"session_id\")\n  s3 = pd.Series(session_month, index = sessions)\n  s3.name = 'session_month'\n  s3 = s3.rename_axis(\"session_id\")\n  s4 = pd.Series(session_year, index = sessions)\n  s4.name = 'session_year'\n  s4 = s4.rename_axis(\"session_id\")\n  return [s1, s2, s3, s4]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates information regarding time and date for each 'session_id'\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\ndate_info = get_date_info(train_filtered)\n\n# Create DataFrame with list of session_ids\ndate_per_sesion_df = pd.concat(date_info, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\ndate_per_sesion_df = date_per_sesion_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\ndate_per_sesion_df = date_per_sesion_df.set_index('session_id')\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the number of events for each session ID:\")\ndate_per_sesion_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Total time and Number of events per session","metadata":{}},{"cell_type":"code","source":"def get_time_and_events_for_each_session(df): \n\n  s1 = df.groupby('session_id')['session_id'].agg('count')\n  s1.name = 'number_of_events'\n\n  s2 = df.groupby('session_id')['elapsed_time2'].agg('max') - df.groupby('session_id')['elapsed_time2'].agg('min')\n  s2.name = 'total_session_time'\n\n  return [s1,s2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of events for each 'session_id'\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"5-12\"]\n\n# Call a function to get the number of events for each event name in the filtered data\ntime_and_events = get_time_and_events_for_each_session(train_filtered)\n\n# Create DataFrame with list of session_ids\nevent_number_per_session_df = pd.concat(time_and_events, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\nevent_number_per_session_df = event_number_per_session_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\nevent_number_per_session_df = event_number_per_session_df.set_index('session_id')\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the number of events for each session ID:\")\nevent_number_per_session_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_unique_texts(df):\n  dfs = []\n  for c in ['text','text_fqid']:\n    tmp = df.groupby('session_id')[c].agg('nunique')\n    tmp.name = tmp.name + '_nunique'\n    dfs.append(tmp)\n  return dfs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of events for `event_name`\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\ntext_unique_count = get_unique_texts(train_filtered)\n\n# Combine the event counts into a single DataFrame\ntexts_feature_df = pd.concat(text_unique_count, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\ntexts_feature_df = texts_feature_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\ntexts_feature_df = texts_feature_df.set_index('session_id')\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\ntexts_feature_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting the data from the x and y coordinates (room)","metadata":{}},{"cell_type":"code","source":"def get_room_coord_data(df):\n  dfs = []\n  for c in ['room_coor_x', 'room_coor_y']:\n    tmp = df.groupby('session_id')[c].agg('mean')\n    tmp.name = tmp.name + '_mean'\n    dfs.append(tmp)  \n  for c in ['room_coor_x', 'room_coor_y']:\n    tmp = df.groupby('session_id')[c].agg('std')\n    tmp.name = tmp.name + '_std'\n    dfs.append(tmp)    \n  return dfs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of events for `event_name`\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\nroom_cord_data = get_room_coord_data(train_filtered)\n\n# Combine the event counts into a single DataFrame\nroom_cord_df = pd.concat(room_cord_data, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\nroom_cord_df = room_cord_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\nroom_cord_df = room_cord_df.set_index('session_id')\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\nroom_cord_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting data from Elapsed Time","metadata":{}},{"cell_type":"code","source":"def get_elapsed_time_dt(df):\n  dfs = []\n\n  tmp = df.groupby('session_id')['elapsed_time2'].agg('mean')\n  tmp.name = tmp.name + '_mean'\n  dfs.append(tmp)\n\n  tmp = df.groupby('session_id')['elapsed_time2'].agg('std')\n  tmp.name = tmp.name + '_std'\n  dfs.append(tmp)  \n\n  return dfs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of eveaor `event_name`\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\nelapsed_data = get_elapsed_time_dt(train_filtered)\n\n# Combine the event counts into a single DataFrame\nelapsed_df = pd.concat(elapsed_data, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\nelapsed_df = elapsed_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\nelapsed_df = elapsed_df.set_index('session_id')\n\nelapsed_df = elapsed_df.fillna(-1)\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\nelapsed_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting data from Page","metadata":{}},{"cell_type":"code","source":"def get_data_page(df):\n  dfs = []\n\n  tmp = df.groupby('session_id')['page'].agg('mean')\n  tmp.name = tmp.name + '_mean'\n  dfs.append(tmp)\n\n  tmp = df.groupby('session_id')['page'].agg('std')\n  tmp.name = tmp.name + '_std'\n  dfs.append(tmp)  \n\n  return dfs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of eveaor `event_name`\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\npage_data = get_data_page(train_filtered)\n\n# Combine the event counts into a single DataFrame\npage_data_df = pd.concat(page_data, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\npage_data_df = page_data_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\npage_data_df = page_data_df.set_index('session_id')\n\npage_data_df = page_data_df.fillna(-1)\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\npage_data_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting data from Level","metadata":{}},{"cell_type":"code","source":"def get_data_level(df):\n  dfs = []\n\n  tmp = df.groupby('session_id')['level'].agg('mean')\n  tmp.name = tmp.name + '_mean'\n  dfs.append(tmp)\n\n  tmp = df.groupby('session_id')['level'].agg('std')\n  tmp.name = tmp.name + '_std'\n  dfs.append(tmp)  \n\n  return dfs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of eveaor `event_name`\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\nlevel_data = get_data_level(train_filtered)\n\n# Combine the event counts into a single DataFrame\nlevel_data_df = pd.concat(level_data, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\nlevel_data_df = level_data_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\nlevel_data_df = level_data_df.set_index('session_id')\n\nlevel_data_df = level_data_df.fillna(-1)\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\nlevel_data_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Counting the number of events for each event name","metadata":{}},{"cell_type":"code","source":"# This function counts the number of events for each event name in a DataFrame\n\ndef get_number_of_events_for_each_event_name(df):\n    # Initialize an empty list to store the counts for each event name\n    event_counts_list = []\n    \n    # Group the DataFrame by session ID and event name, and count the number of events for each group\n    grouped_counts = df.groupby(by=[\"session_id\", \"event_name\"])[\"index\"].count()\n    \n    # Get a list of unique event names in the DataFrame\n    event_names = grouped_counts.index.get_level_values(1).unique()\n    \n    # Iterate through each event name and extract the counts for that name\n    for event_name in event_names:\n        counts_for_event_name = grouped_counts.loc[:, event_name]\n        \n        # Rename the column to include the event name\n        counts_for_event_name = counts_for_event_name.rename(f\"{event_name}_count\")\n        \n        # Add the counts to the list of event counts\n        event_counts_list.append(counts_for_event_name)\n    \n    # Return the list of event counts\n    return event_counts_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of events for `event_name`\")\n\n# Select a subset of data from the original training data based on a filter condition\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call a function to get the number of events for each event name in the filtered data\nevent_counts_series = get_number_of_events_for_each_event_name(train_filtered)\n\n# Combine the event counts into a single DataFrame\nevent_counts_per_event_name_df = pd.concat(event_counts_series, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\nevent_counts_per_event_name_df = event_counts_per_event_name_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\nevent_counts_per_event_name_df = event_counts_per_event_name_df.set_index('session_id')\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\nevent_counts_per_event_name_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Counting the number of events for each level ","metadata":{}},{"cell_type":"code","source":"# This function counts the number of events for each level in a DataFrame\n\ndef get_number_of_events_for_each_level(df):\n    # Initialize an empty list to store the counts for each level\n    level_counts_list = []\n    \n    # Group the DataFrame by session ID and level, and count the number of events for each group\n    grouped_counts = df.groupby(by=[\"session_id\", \"level\"])[\"index\"].count()\n    \n    # Get a list of unique levels in the DataFrame\n    levels = grouped_counts.index.get_level_values(1).unique()\n    \n    # Iterate through each level and extract the counts for that level\n    for level in levels:\n        counts_for_level = grouped_counts.loc[:, level]\n        \n        # Rename the column to include the level number\n        counts_for_level = counts_for_level.rename(f\"level{level}_count\")\n        \n        # Add the counts to the list of level counts\n        level_counts_list.append(counts_for_level)\n    \n    # Return the list of level counts\n    return level_counts_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function generates the number of events for `level`\")\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\nevent_counts_series = get_number_of_events_for_each_level(train_filtered)\n\n# Combine the event counts into a single DataFrame\nevent_counts_per_level_df = pd.concat(event_counts_series, axis=1)\n\n# Reset the index of the DataFrame to make the session ID a column\nevent_counts_per_level_df = event_counts_per_level_df.reset_index()\n\n# Set the session ID as the index of the DataFrame\nevent_counts_per_level_df = event_counts_per_level_df.set_index('session_id')\n\n# Display the first few rows of the DataFrame\nprint(\"Here are the event counts for each session ID:\")\nevent_counts_per_level_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bringing together the data of the \"hover\" variable ","metadata":{}},{"cell_type":"code","source":"# This function calculates aggregate statistics for the \"hover_duration\" column in a DataFrame\n\ndef get_hover_duration_aggregate_data(df):\n    # Initialize an empty list to store the results for each aggregate statistic\n    results_list = []\n    \n    # Calculate the sum of hover durations for each session\n    hover_duration_sum = df.groupby(by=[\"session_id\"])[\"hover_duration\"].sum()\n    hover_duration_sum = hover_duration_sum.rename(\"hover_duration_sum\")\n    results_list.append(hover_duration_sum)\n    \n    # Calculate the mean of hover durations for each session\n    hover_duration_mean = df.groupby(by=[\"session_id\"])[\"hover_duration\"].mean()\n    hover_duration_mean = hover_duration_mean.rename(\"hover_duration_mean\")\n    results_list.append(hover_duration_mean)\n    \n    # Calculate the standard deviation of hover durations for each session\n    hover_duration_std = df.groupby(by=[\"session_id\"])[\"hover_duration\"].std()\n    hover_duration_std = hover_duration_std.rename(\"hover_duration_std\")\n    results_list.append(hover_duration_std)\n    \n    # Calculate the count of hover durations for each session\n    hover_duration_count = df.groupby(by=[\"session_id\"])[\"hover_duration\"].count()\n    hover_duration_count = hover_duration_count.rename(\"hover_duration_count\")\n    results_list.append(hover_duration_count)\n    \n    # Return the list of results\n    return results_list\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nThis function calculates the sum, mean, standard deviation, and count of hover durations.\")\n\n# Filter the DataFrame to include only rows where \"level_group\" is \"0-4\"\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]\n\n# Call the function to calculate the aggregate statistics and store the results in a list of DataFrames\nresults_series = get_hover_duration_aggregate_data(train_filtered)\n\n# Concatenate the DataFrames in the list into a single DataFrame, reset the index, and set the index to \"session_id\"\nresults_df = pd.concat(results_series, axis=1)\nresults_df = results_df.reset_index()\nresults_df = results_df.set_index('session_id')\n\nprint(\"Here are the event counts for each session ID:\")\nresults_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bringing together all the game configurations \n","metadata":{}},{"cell_type":"code","source":"# Define a function to extract game configuration data from a DataFrame\ndef get_game_config_data(df):\n    feature_dfs = []  # Create an empty list to hold DataFrames for each feature\n    # Loop over each game configuration option of interest\n    for config_option in [\"fullscreen\", \"hq\", \"music\"]:\n        # Group the DataFrame by session ID and extract the first value of the configuration option for each session\n        config_values = df.groupby(by=[\"session_id\"])[config_option].first()\n        config_values = config_values.rename(config_option)  # Rename the resulting Series to match the configuration option\n        feature_dfs.append(config_values)  # Append the Series to the list of feature DataFrames\n    return feature_dfs  # Return the list of feature DataFrames","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract game configuration features for a subset of the training data\ntrain_filtered = train_original[train_original[\"level_group\"] == \"0-4\"]  # Filter the training data to a specific level group of interest\nconfig_features = get_game_config_data(train_filtered)  # Call the function to extract game configuration features\nconfig_features_df = pd.concat(config_features, axis=1)  # Concatenate the list of feature DataFrames into a single DataFrame\nconfig_features_df = config_features_df.reset_index()  # Reset the index of the DataFrame to include the session IDs\nconfig_features_df = config_features_df.set_index('session_id')  # Set the index of the DataFrame to be the session IDs\nconfig_features_df.head() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bringing together all the new features","metadata":{}},{"cell_type":"code","source":"def concat_features(df):\n    dfs1 = get_number_of_events_for_each_event_name(df) #For each event name (Ex: checkpoint, cutscene, mapclick), get number of appearences for session\n    dfs2 = get_number_of_events_for_each_level(df) #For each level, get number of events/actions that occured in it\n    dfs3 = get_hover_duration_aggregate_data(df) #Calculates some statistical data about the column 'hover', like its sum, mean, std, for each session\n    dfs4 = get_game_config_data(df) #Simple yes or no column about the configurations fullscreen, hq and music\n\n    dfs5 = get_unique_texts(df) #Getting number of unique texts that appeared, both for text and text_fqid, per session\n    dfs6 = get_room_coord_data(df) #Getting statistical data both from the x and y room coordinates, like before (mean, std), for each session\n    dfs7 = get_elapsed_time_dt(df) #Getting statistical data from the (new and fixed) elapsed_time column for each session\n    dfs8 = get_data_page(df) #Getting statistical data from the page column (mean, std)\n    dfs9 = get_data_level(df) #Getting statistical data from the level column (mean, std)\n        \n    dfs10 = get_time_and_events_for_each_session(df) #Getting the number of events and total time for each session\n    dfs11 = get_date_info(df) # Getting info about the date and hour of when the first event of each session in the dataframe started (hour, day of the week, month, year)\n\n    dfs_total = (dfs1 + dfs2 + dfs3 + dfs4 + dfs10)\n\n    independent_variables = pd.concat(dfs_total, axis = 1)\n    independent_variables = independent_variables.fillna(-1) #fill NaN with -1. Done specially for data regarding room coordinates and page, since there are a lot missing\n    independent_variables = independent_variables.reset_index()\n    independent_variables = independent_variables.set_index('session_id')\n\n    return independent_variables","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can split the data by group of questions since the submission API gives us level segments 'level_group'.","metadata":{}},{"cell_type":"code","source":"early_game_questions = train_original[train_original[\"level_group\"] == \"0-4\"]\nEARLY_FEATURES = concat_features(early_game_questions)\ndel early_game_questions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mid_game_questions = train_original[train_original[\"level_group\"] == \"5-12\"]\nMID_FEATURES = concat_features(mid_game_questions)\ndel mid_game_questions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"late_game_questions = train_original[train_original[\"level_group\"] == \"13-22\"]\nLATE_FEATURES = concat_features(late_game_questions)\ndel late_game_questions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EARLY_FEATURES.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dependent_variable_matrix(df):\n    df[\"question\"] = df[\"session_id\"].str.split(\"_\").str[1].str[1:].astype(int)\n    df[\"session\"] = df[\"session_id\"].str.split(\"_\").str[0]\n    return df\n    \nDEPENDENT_VARIABLES = get_dependent_variable_matrix(train_labels)\nDEPENDENT_VARIABLES","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the model","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nfrom sklearn.model_selection import GridSearchCV\nimport xgboost as xgb\n\n# Set the model parameters for grid search\nparameters = {\n    'max_depth': [3, 4], \n    'n_estimators': [20, 50, 100],\n    'learning_rate': [0.01, 0.05, 0.1],\n    'max_depth': [1, 2]\n}\n\n# Initialize dictionaries to store models and f1 scores\nMODELS = {}\nF1 = {}\n\n# Loop through each question and train a model\nfor question in range(1, 19):\n    print(f'\\nTRAIN QUESTION {question} MODEL')\n    \n    # Set X features based on the current question\n    if question <= 3:\n        X = EARLY_FEATURES\n    elif question <= 13:\n        X = MID_FEATURES\n    elif question <= 18:\n        X = LATE_FEATURES\n    \n    # Set y target based on the current question\n    y = DEPENDENT_VARIABLES[DEPENDENT_VARIABLES[\"question\"] == question][\"correct\"]\n    \n    # Initialize a new XGBoost classifier and perform grid search to find the best hyperparameters\n    model_xgb = xgb.XGBClassifier(random_state = 1)\n    model_xgb = GridSearchCV(\n        model_xgb, \n        parameters, \n        cv=3,\n        scoring='f1')\n    model_xgb.fit(X, y)\n    \n    # Store the model and its f1 score in their respective dictionaries\n    MODELS[f\"question {question} model\"] = model_xgb\n    F1[f\"question {question} f1\"] = model_xgb.best_score_\n    \n    # Print the best f1 score and hyperparameters for the current question\n    print(f\"\\tf1 score is {model_xgb.best_score_:.3f}\")\n    print(f\"\\tbest params are {model_xgb.best_params_}\")\n    print(f'QUESTION {question} MODEL COMPLETE')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}