{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/predict-student-performance-from-game-play'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-06T05:07:29.111782Z","iopub.execute_input":"2023-07-06T05:07:29.112431Z","iopub.status.idle":"2023-07-06T05:07:29.128704Z","shell.execute_reply.started":"2023-07-06T05:07:29.112393Z","shell.execute_reply":"2023-07-06T05:07:29.127724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Student Performance from Game Play\n","metadata":{}},{"cell_type":"code","source":"with open('/kaggle/input/predict-student-performance-from-game-play/jo_wilder_310/__init__.py', 'r') as file:\n    code = file.read()\n\nprint(code)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:07:29.130212Z","iopub.execute_input":"2023-07-06T05:07:29.130637Z","iopub.status.idle":"2023-07-06T05:07:29.138695Z","shell.execute_reply.started":"2023-07-06T05:07:29.130615Z","shell.execute_reply":"2023-07-06T05:07:29.137500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd     \nimport seaborn as sns \nimport math\nimport statistics\nimport missingno as msno\nimport ast\n\nimport matplotlib.pyplot as plt\nimport gc\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:07:29.139901Z","iopub.execute_input":"2023-07-06T05:07:29.140496Z","iopub.status.idle":"2023-07-06T05:07:29.147256Z","shell.execute_reply.started":"2023-07-06T05:07:29.140468Z","shell.execute_reply":"2023-07-06T05:07:29.146556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Dataset","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\nimport numpy as np\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndf = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\n\nprint(\"Full train dataset shape is {}\".format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:07:29.148879Z","iopub.execute_input":"2023-07-06T05:07:29.149533Z","iopub.status.idle":"2023-07-06T05:08:53.479993Z","shell.execute_reply.started":"2023-07-06T05:07:29.149505Z","shell.execute_reply":"2023-07-06T05:08:53.476457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature engineer \nthe data is imbalanced and it's big, so that could be a bad thing or a good thing, the key to unlocking imbalanced data, is not the  ML and NN but good data analysis if you can capture the pattern that causes that event if you do that this could be a great chance to achieve high F1-score without the need to complicated  model .\n","metadata":{}},{"cell_type":"markdown","source":"# data analysis \n1)pca_data:\nThis data is numerical and I use Principal component analysis to  dimension reducing  \n2)in sup_data and mean_data:\nI grouped them  by 'session_id' and 'level_group' and then aggregate the data into a list, the goal of this move is to save the order     \nI did some exploration on the data and find that the data vary by the difference in the length of this list, and if you take the length without the repeat of adjacent events that can lead to a way of Distinguish the data    \n\nfor the event_name it looks a promising way to better understand the data so I did the levenshtein_distance_vec  algorithm on them and that was a great idea, it's a way to compare the similarity of the path  between two lists the great thing about this algorithm is that keeps an eye on the order of how that event takes place, the paths that I use as an argument is a path that students leave behind after  answering all questions right   \n\nfor elapsed_time and hover_duration I use the mean and the std, sum\n\nthe students who have high scores had less time \n \n","metadata":{}},{"cell_type":"code","source":"pca_data = df[['session_id', 'level_group', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y']]\n\nmean_data = df[['session_id', 'level_group','elapsed_time', 'event_name', 'name']]\n\nsup_data = df[['session_id',  'level_group','index','page','hover_duration', 'text', 'fqid', 'room_fqid', 'text_fqid']]\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:53.490245Z","iopub.execute_input":"2023-07-06T05:08:53.491901Z","iopub.status.idle":"2023-07-06T05:08:54.136074Z","shell.execute_reply.started":"2023-07-06T05:08:53.491846Z","shell.execute_reply":"2023-07-06T05:08:54.134411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport psutil\n\n# Check memory usage before deleting the DataFrame\nbefore_memory = psutil.virtual_memory().used\n\n# Delete the DataFrame\ndel df\n\n# Trigger an immediate garbage collection\ngc.collect()\n\n# Check memory usage after deleting the DataFrame and garbage collection\nafter_memory = psutil.virtual_memory().used\n\n# Calculate the memory freed\nmemory_freed = before_memory - after_memory\nprint(f\"Memory freed: {memory_freed} bytes\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:54.138460Z","iopub.execute_input":"2023-07-06T05:08:54.138981Z","iopub.status.idle":"2023-07-06T05:08:54.539748Z","shell.execute_reply.started":"2023-07-06T05:08:54.138938Z","shell.execute_reply":"2023-07-06T05:08:54.538146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n# ADD EXTRA COLUMNS\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:54.542272Z","iopub.execute_input":"2023-07-06T05:08:54.543603Z","iopub.status.idle":"2023-07-06T05:08:55.991393Z","shell.execute_reply.started":"2023-07-06T05:08:54.543518Z","shell.execute_reply":"2023-07-06T05:08:55.989653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test0 = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:55.992755Z","iopub.execute_input":"2023-07-06T05:08:55.993017Z","iopub.status.idle":"2023-07-06T05:08:56.022235Z","shell.execute_reply.started":"2023-07-06T05:08:55.992995Z","shell.execute_reply":"2023-07-06T05:08:56.020421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df= df.drop(['fullscreen', 'hq', 'music'], axis=1)\n    \nz_scores = np.abs((mean_data['elapsed_time'] - mean_data['elapsed_time'].mean()) / mean_data['elapsed_time'].std())\noutliers = mean_data[z_scores > 3]\nmean_data.loc[outliers.index, 'elapsed_time'] = mean_data['elapsed_time'].mean()\n\nz_scores = np.abs((sup_data['hover_duration'] - sup_data['hover_duration'].mean()) / sup_data['hover_duration'].std())\noutliers = sup_data[z_scores > 3]\nsup_data.loc[outliers.index, 'hover_duration'] = sup_data['hover_duration'].mean()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:56.024447Z","iopub.execute_input":"2023-07-06T05:08:56.024966Z","iopub.status.idle":"2023-07-06T05:08:56.848841Z","shell.execute_reply.started":"2023-07-06T05:08:56.024938Z","shell.execute_reply":"2023-07-06T05:08:56.846958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nimport numpy as np\n\n# Create an instance of PCA with 2 components\npca = PCA(n_components=1)\n\npca_df = pca_data.groupby(['session_id', 'level_group']).mean()\ncoor_pca = pca.fit_transform(pca_df.iloc[:,2:])\ncoor_data = pd.DataFrame(coor_pca)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:56.854529Z","iopub.execute_input":"2023-07-06T05:08:56.856702Z","iopub.status.idle":"2023-07-06T05:08:58.216339Z","shell.execute_reply.started":"2023-07-06T05:08:56.856643Z","shell.execute_reply":"2023-07-06T05:08:58.215397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\ndef dcd(x):\n    i = 0\n    #x = ast.literal_eval(x)\n    \n    if isinstance(x, list):\n        while i < len(x) - 1:\n            if x[i] == x[i + 1]:\n                del x[i]\n            else:\n                i += 1\n        x = np.array(x)\n        x = x.flatten()\n        x = list(x)\n    else:\n        x = [x]  # Handle non-list values\n        \n    return x\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:58.217624Z","iopub.execute_input":"2023-07-06T05:08:58.217874Z","iopub.status.idle":"2023-07-06T05:08:58.227791Z","shell.execute_reply.started":"2023-07-06T05:08:58.217854Z","shell.execute_reply":"2023-07-06T05:08:58.226296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport numpy as np\n\ndef levenshtein_distance_vec(path1, path2):\n    m = len(path1)\n    n = len(path2)\n\n    # Create a matrix to store the edit distances\n    distance = np.zeros((m+1, n+1))\n\n    # Initialize the first row and column of the matrix\n    distance[:, 0] = np.arange(m+1)\n    distance[0, :] = np.arange(n+1)\n\n    # Calculate the edit distance using vectorization\n    for i in range(1, m+1):\n        for j in range(1, n+1):\n            cost = (path1[i-1] != path2[j-1])\n            distance[i, j] = min(distance[i-1, j] + 1, distance[i, j-1] + 1, distance[i-1, j-1] + cost)\n\n    return distance[m, n]\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:58.229006Z","iopub.execute_input":"2023-07-06T05:08:58.229295Z","iopub.status.idle":"2023-07-06T05:08:58.243949Z","shell.execute_reply.started":"2023-07-06T05:08:58.229273Z","shell.execute_reply":"2023-07-06T05:08:58.242584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sup_data = sup_data.groupby(['session_id', 'level_group']).agg(list).reset_index()\n\nsup_data['index'] = sup_data['index'].apply(len)\nsup_data['page'] = sup_data['page'].apply(lambda x: np.nansum(np.array(x)))\nsup_data['hover_duration'] = sup_data['hover_duration'].apply(lambda x: np.nansum(np.array(x)))\n\ndef remove_consecutive_duplicates(x):\n    return [x[i] for i in range(len(x)) if i == 0 or x[i] != x[i-1]]\n\ncolumns_to_process = ['text', 'room_fqid', 'text_fqid', 'fqid']\nfor column in columns_to_process:\n    sup_data[column] = sup_data[column].apply(remove_consecutive_duplicates)\n    sup_data[column] = sup_data[column].apply(len)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:08:58.248875Z","iopub.execute_input":"2023-07-06T05:08:58.250976Z","iopub.status.idle":"2023-07-06T05:10:12.222357Z","shell.execute_reply.started":"2023-07-06T05:08:58.250938Z","shell.execute_reply":"2023-07-06T05:10:12.221182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_data = mean_data.groupby(['session_id', 'level_group']).agg(list).reset_index()\nmean_data['mean_elapsed_time'] = mean_data['elapsed_time'].apply(lambda x: np.mean(x))\nmean_data['var_elapsed_time'] = mean_data['elapsed_time'].apply(lambda x: np.var(x))\nmean_data['elapsed_time'] = mean_data['elapsed_time'].apply(sum)\n\nmean_data['event_name'] = mean_data['event_name'].apply(remove_consecutive_duplicates)\nmean_data['name'] = mean_data['name'].apply(remove_consecutive_duplicates)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:10:12.227119Z","iopub.execute_input":"2023-07-06T05:10:12.227752Z","iopub.status.idle":"2023-07-06T05:10:44.660127Z","shell.execute_reply.started":"2023-07-06T05:10:12.227719Z","shell.execute_reply":"2023-07-06T05:10:44.659008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_mark = ['cutscene_click',\n 'person_click',\n 'navigate_click',\n 'notification_click',\n 'object_click',\n 'navigate_click',\n 'notification_click',\n 'object_click',\n 'navigate_click',\n 'notification_click',\n 'object_click',\n 'navigate_click',\n 'cutscene_click',\n 'object_click',\n 'cutscene_click',\n 'navigate_click',\n 'object_click',\n 'navigate_click',\n 'object_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'notification_click',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_click',\n 'navigate_click',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'cutscene_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'checkpoint']\nfull_mark1 = ['navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'observation_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'observation_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'notebook_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'observation_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'observation_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'map_hover',\n 'map_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'notification_click',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_hover',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'checkpoint']\n\nfull_mark2 = ['navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'cutscene_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'observation_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'notification_click',\n 'object_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'person_click',\n 'navigate_click',\n 'object_hover',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'object_hover',\n 'object_click',\n 'notification_click',\n 'object_click',\n 'notification_click',\n 'object_hover',\n 'object_click',\n 'navigate_click',\n 'map_click',\n 'map_hover',\n 'map_click',\n 'navigate_click',\n 'checkpoint']\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-07-06T05:10:44.669725Z","iopub.execute_input":"2023-07-06T05:10:44.670270Z","iopub.status.idle":"2023-07-06T05:10:44.683742Z","shell.execute_reply.started":"2023-07-06T05:10:44.670242Z","shell.execute_reply":"2023-07-06T05:10:44.682484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the custom function to calculate edit distance for each type\nmean_data.loc[::3, 'event_name_distance'] = mean_data.loc[::3, 'event_name'].apply(lambda x: levenshtein_distance_vec(x, full_mark))\nmean_data.loc[1::3, 'event_name_distance'] = mean_data.loc[1::3, 'event_name'].apply(lambda x: levenshtein_distance_vec(x, full_mark1))\nmean_data.loc[2::3, 'event_name_distance'] = mean_data.loc[2::3, 'event_name'].apply(lambda x: levenshtein_distance_vec(x, full_mark2))\nmean_data['event_name'] = mean_data['event_name'].apply(len)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:10:44.685931Z","iopub.execute_input":"2023-07-06T05:10:44.686278Z","iopub.status.idle":"2023-07-06T05:23:17.327554Z","shell.execute_reply.started":"2023-07-06T05:10:44.686249Z","shell.execute_reply":"2023-07-06T05:23:17.326532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_data['name'] = mean_data['name'].apply(len)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.331355Z","iopub.execute_input":"2023-07-06T05:23:17.331706Z","iopub.status.idle":"2023-07-06T05:23:17.389017Z","shell.execute_reply.started":"2023-07-06T05:23:17.331685Z","shell.execute_reply":"2023-07-06T05:23:17.388317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coor_data = coor_data.rename(columns={0: 'coor_0'})\n#coor_data = coor_data.rename(columns={1: 'coor_1'})\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.390668Z","iopub.execute_input":"2023-07-06T05:23:17.391091Z","iopub.status.idle":"2023-07-06T05:23:17.396925Z","shell.execute_reply.started":"2023-07-06T05:23:17.391067Z","shell.execute_reply":"2023-07-06T05:23:17.395527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#merged_data = pd.concat([mean_data, sup_data.iloc[:,2:], coor_data.iloc[:,2:]], axis=0)\nmerged_df = mean_data.join(sup_data.iloc[:,2:], how='inner')\n\n# Join the third dataset with the merged_df based on the index\ndataset_df = merged_df.join(coor_data, how='inner')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.398790Z","iopub.execute_input":"2023-07-06T05:23:17.399138Z","iopub.status.idle":"2023-07-06T05:23:17.420786Z","shell.execute_reply.started":"2023-07-06T05:23:17.399104Z","shell.execute_reply":"2023-07-06T05:23:17.419449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = dataset_df.set_index('session_id')\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.422561Z","iopub.execute_input":"2023-07-06T05:23:17.422853Z","iopub.status.idle":"2023-07-06T05:23:17.433960Z","shell.execute_reply.started":"2023-07-06T05:23:17.422828Z","shell.execute_reply":"2023-07-06T05:23:17.433254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(df):\n    pca_data = df[['session_id', 'level_group', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y']]\n    mean_data = df[['session_id', 'level_group', 'elapsed_time', 'event_name', 'name']]\n    sup_data = df[['session_id', 'level_group', 'index', 'page', 'hover_duration', 'text', 'fqid', 'room_fqid', 'text_fqid']]\n\n    z_scores = np.abs((mean_data['elapsed_time'] - mean_data['elapsed_time'].mean()) / mean_data['elapsed_time'].std())\n    outliers = mean_data[z_scores > 3]\n    mean_data.loc[outliers.index, 'elapsed_time'] = mean_data['elapsed_time'].mean()\n\n    z_scores = np.abs((sup_data['hover_duration'] - sup_data['hover_duration'].mean()) / sup_data['hover_duration'].std())\n    outliers = sup_data[z_scores > 3]\n    sup_data.loc[outliers.index, 'hover_duration'] = sup_data['hover_duration'].mean()\n\n    from sklearn.decomposition import PCA\n\n    pca1 = PCA(n_components=1)\n\n    pca_df = pca_data.groupby(['session_id', 'level_group']).mean()\n    coor_pca = pca1.fit_transform(pca_df.iloc[:,2:])\n    coor_data = pd.DataFrame(coor_pca)\n\n    def levenshtein_distance_vec(path1, path2):\n        m = len(path1)\n        n = len(path2)\n        distance = np.zeros((m + 1, n + 1))\n        distance[:, 0] = np.arange(m + 1)\n        distance[0, :] = np.arange(n + 1)\n        for i in range(1, m + 1):\n            for j in range(1, n + 1):\n                cost = (path1[i - 1] != path2[j - 1])\n                distance[i, j] = min(distance[i - 1, j] + 1, distance[i, j - 1] + 1, distance[i - 1, j - 1] + cost)\n        return distance[m, n]\n\n    sup_data = sup_data.groupby(['session_id', 'level_group']).agg(list).reset_index()\n    sup_data['index'] = sup_data['index'].apply(len)\n    sup_data['page'] = sup_data['page'].apply(lambda x: np.nansum(np.array(x)))\n    sup_data['hover_duration'] = sup_data['hover_duration'].apply(lambda x: np.nansum(np.array(x)))\n\n    def remove_consecutive_duplicates(x):\n        return [x[i] for i in range(len(x)) if i == 0 or x[i] != x[i - 1]]\n\n    columns_to_process = ['text', 'room_fqid', 'text_fqid', 'fqid']\n    for column in columns_to_process:\n        sup_data[column] = sup_data[column].apply(remove_consecutive_duplicates)\n        sup_data[column] = sup_data[column].apply(len)\n\n    mean_data = mean_data.groupby(['session_id', 'level_group']).agg(list).reset_index()\n    mean_data['mean_elapsed_time'] = mean_data['elapsed_time'].apply(lambda x: np.mean(x))\n    mean_data['var_elapsed_time'] = mean_data['elapsed_time'].apply(lambda x: np.var(x))\n    mean_data['elapsed_time'] = mean_data['elapsed_time'].apply(sum)\n    mean_data['event_name'] = mean_data['event_name'].apply(remove_consecutive_duplicates)\n    mean_data['name'] = mean_data['name'].apply(remove_consecutive_duplicates)\n\n    full_mark = ['cutscene_click', 'person_click', 'navigate_click', 'notification_click', 'object_click', 'navigate_click',\n                 'notification_click', 'object_click', 'navigate_click', 'notification_click', 'object_click', 'navigate_click',\n                 'cutscene_click', 'object_click', 'cutscene_click', 'navigate_click', 'object_click', 'navigate_click',\n                 'object_click', 'navigate_click', 'cutscene_click', 'navigate_click', 'notification_click', 'object_click',\n                 'notification_click', 'object_hover', 'object_click', 'navigate_click', 'person_click', 'navigate_click',\n                 'map_click', 'navigate_click', 'object_click', 'notification_click', 'object_hover', 'object_click',\n                 'cutscene_click', 'navigate_click', 'map_hover', 'map_click', 'navigate_click', 'checkpoint']\n\n    full_mark1 = ['navigate_click', 'map_hover', 'map_click', 'navigate_click', 'cutscene_click', 'navigate_click',\n                  'cutscene_click', 'navigate_click', 'observation_click', 'navigate_click', 'person_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'person_click', 'navigate_click', 'object_click', 'notification_click',\n                  'object_hover', 'object_click', 'navigate_click', 'observation_click', 'navigate_click', 'cutscene_click',\n                  'navigate_click', 'cutscene_click', 'navigate_click', 'cutscene_click', 'navigate_click', 'notebook_click',\n                  'navigate_click', 'person_click', 'navigate_click', 'cutscene_click', 'navigate_click', 'person_click',\n                  'navigate_click', 'person_click', 'navigate_click', 'map_hover', 'map_click', 'navigate_click',\n                  'observation_click', 'navigate_click', 'person_click', 'navigate_click', 'person_click', 'navigate_click',\n                  'observation_click', 'navigate_click', 'person_click', 'navigate_click', 'object_hover', 'object_click',\n                  'notification_click', 'object_click', 'object_hover', 'object_click', 'navigate_click', 'person_click',\n                  'navigate_click', 'map_hover', 'map_click', 'map_hover', 'map_click', 'map_hover', 'map_click',\n                  'navigate_click', 'person_click', 'navigate_click', 'notification_click', 'object_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'map_hover', 'map_click', 'navigate_click', 'person_click',\n                  'navigate_click', 'object_click', 'object_hover', 'object_click', 'notification_click', 'object_hover',\n                  'object_click', 'navigate_click', 'person_click', 'navigate_click', 'map_hover', 'map_click',\n                  'navigate_click', 'person_click', 'navigate_click', 'object_hover', 'object_click', 'object_hover',\n                  'object_click', 'object_hover', 'object_click', 'notification_click', 'object_hover', 'object_click',\n                  'navigate_click', 'map_hover', 'map_click', 'navigate_click', 'checkpoint']\n\n    full_mark2 = ['navigate_click', 'map_hover', 'map_click', 'navigate_click', 'cutscene_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'cutscene_click', 'navigate_click', 'cutscene_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'person_click', 'navigate_click', 'observation_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'map_click', 'map_hover', 'map_click', 'navigate_click', 'person_click',\n                  'navigate_click', 'object_click', 'object_hover', 'object_click', 'notification_click', 'object_hover',\n                  'object_click', 'navigate_click', 'object_click', 'object_hover', 'object_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'map_click', 'navigate_click', 'person_click', 'navigate_click',\n                  'object_click', 'object_hover', 'object_click', 'notification_click', 'object_hover', 'object_click',\n                  'navigate_click', 'person_click', 'navigate_click', 'map_click', 'navigate_click', 'person_click',\n                  'navigate_click', 'object_click', 'object_hover', 'object_click', 'notification_click', 'object_click',\n                  'object_hover', 'object_click', 'navigate_click', 'notification_click', 'object_click', 'navigate_click',\n                  'person_click', 'navigate_click', 'person_click', 'navigate_click', 'map_hover', 'map_click',\n                  'navigate_click', 'person_click', 'navigate_click', 'person_click', 'navigate_click', 'object_hover',\n                  'object_click', 'object_hover', 'object_click', 'object_hover', 'object_click', 'notification_click',\n                  'object_click', 'notification_click', 'object_hover', 'object_click', 'navigate_click', 'map_click',\n                  'map_hover', 'map_click', 'navigate_click', 'checkpoint']\n\n    # Apply the custom function to calculate edit distance for each type\n    mean_data.loc[::3, 'event_name_distance'] = mean_data.loc[::3, 'event_name'].apply(lambda x: levenshtein_distance_vec(x, full_mark))\n    mean_data.loc[1::3, 'event_name_distance'] = mean_data.loc[1::3, 'event_name'].apply(lambda x: levenshtein_distance_vec(x, full_mark1))\n    mean_data.loc[2::3, 'event_name_distance'] = mean_data.loc[2::3, 'event_name'].apply(lambda x: levenshtein_distance_vec(x, full_mark2))\n    mean_data['event_name'] = mean_data['event_name'].apply(len)\n    mean_data['name'] = mean_data['name'].apply(len)\n    #coor_data = coor_data.rename(columns={0: 'coor_0', 1: 'coor_1'})\n    coor_data = coor_data.rename(columns={0: 'coor_0'})\n\n        #merged_data = pd.concat([mean_data, sup_data.iloc[:,2:], coor_data.iloc[:,2:]], axis=0)\n    merged_df = mean_data.join(sup_data.iloc[:,2:], how='inner')\n\n    # Join the third dataset with the merged_df based on the index\n    dataset_df = merged_df.join(coor_data, how='inner')\n    data = dataset_df.set_index('session_id')\n    return data\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.435176Z","iopub.execute_input":"2023-07-06T05:23:17.435565Z","iopub.status.idle":"2023-07-06T05:23:17.462178Z","shell.execute_reply.started":"2023-07-06T05:23:17.435544Z","shell.execute_reply":"2023-07-06T05:23:17.460401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.463817Z","iopub.execute_input":"2023-07-06T05:23:17.464102Z","iopub.status.idle":"2023-07-06T05:23:17.530330Z","shell.execute_reply.started":"2023-07-06T05:23:17.464078Z","shell.execute_reply":"2023-07-06T05:23:17.529261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training cat boost model \n","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nfrom sklearn.metrics import accuracy_score, f1_score\n\n# Fetch the unique list of user sessions in the validation dataset\nVALID_USER_LIST = valid_x.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST), 18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question\nevaluation_dict = {}\n\n# Iterate through questions 1 to 18 to train models for each question, evaluate the trained model, and store the predicted values.\nfor q_no in range(1, 19):\n    # Select level group for the question based on the q_no.\n    if q_no <= 3:\n        grp = '0-4'\n    elif q_no <= 13:\n        grp = '5-12'\n    elif q_no <= 22:\n        grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n\n    # Filter the rows in the datasets based on the selected level group.\n    train_df = train_x.loc[train_x.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp]\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = labels.loc[labels.q == q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q == q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # Convert the dataset from Pandas format into numpy arrays.\n    train_data = train_df.drop(columns=['level_group', 'correct']).values\n    train_labels = train_df[\"correct\"].values\n    valid_data = valid_df.drop(columns=['level_group', 'correct']).values\n    valid_labels = valid_df[\"correct\"].values\n\n    # Create the CatBoostClassifier model.\n    catboost_model = CatBoostClassifier(verbose=0)\n\n    # Train the model.\n    catboost_model.fit(train_data, train_labels)\n\n    # Store the model.\n    models[f'{grp}_{q_no}'] = catboost_model\n\n    # Evaluate the trained model on the validation dataset and store the evaluation accuracy and F1 score in the `evaluation_dict`.\n    predictions = catboost_model.predict(valid_data)\n    accuracy = accuracy_score(valid_labels, predictions)\n    f1 = f1_score(valid_labels, predictions)\n    evaluation_dict[q_no] = {'accuracy': accuracy, 'f1_score': f1}\n\n    # Use the trained model to make predictions on the validation dataset and store the predicted values in the `prediction_df` dataframe.\n    prediction_df.loc[valid_users, q_no-1] = predictions.flatten()\n\n# Iterate over the evaluation dictionary and print the accuracy and F1 score for each question\nfor q_no, scores in evaluation_dict.items():\n    accuracy = scores['accuracy']\n    f1 = scores['f1_score']\n    print(f\"Question {q_no}: Accuracy {accuracy:.4f}, F1 Score: {f1:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:23:17.531797Z","iopub.execute_input":"2023-07-06T05:23:17.532057Z","iopub.status.idle":"2023-07-06T05:25:09.552280Z","shell.execute_reply.started":"2023-07-06T05:23:17.532035Z","shell.execute_reply":"2023-07-06T05:25:09.551413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inspect the Accuracy of the models.¶\n","metadata":{}},{"cell_type":"code","source":"# Iterate over the evaluation dictionary and print the accuracy and F1 score for each question\nfor q_no, scores in evaluation_dict.items():\n    accuracy = scores['accuracy']\n    f1 = scores['f1_score']\n    print(f\"Question {q_no}: Accuracy {accuracy:.4f}, F1 Score: {f1:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:09.553501Z","iopub.execute_input":"2023-07-06T05:25:09.553934Z","iopub.status.idle":"2023-07-06T05:25:09.559280Z","shell.execute_reply.started":"2023-07-06T05:25:09.553905Z","shell.execute_reply":"2023-07-06T05:25:09.558456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_accuracy = np.mean([scores['accuracy'] for scores in evaluation_dict.values()])\nmean_f1_score = np.mean([scores['f1_score'] for scores in evaluation_dict.values()])\n\nprint(f\"Mean Accuracy: {mean_accuracy:.4f}\")\nprint(f\"Mean F1 Score: {mean_f1_score:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:09.560463Z","iopub.execute_input":"2023-07-06T05:25:09.561574Z","iopub.status.idle":"2023-07-06T05:25:09.577854Z","shell.execute_reply.started":"2023-07-06T05:25:09.561507Z","shell.execute_reply":"2023-07-06T05:25:09.576142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize the model¶\n","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nimport matplotlib.pyplot as plt\n\n# Iterate over the models dictionary\nfor key, model in models.items():\n    # Extract the level group and question number from the key\n    level_group, q_no = key.split('_')\n\n    # Visualize the decision tree using the plot_tree() function\n    tree_index = 0  # Index of the tree you want to visualize (0 for the first tree)\n    plot = CatBoostClassifier.plot_tree(\n        model,\n        tree_idx=tree_index,\n        pool=train_data  # Provide the training data used to build the model\n    )\n\nplot#-----------> for the last model q18\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:09.579426Z","iopub.execute_input":"2023-07-06T05:25:09.579991Z","iopub.status.idle":"2023-07-06T05:25:09.905501Z","shell.execute_reply.started":"2023-07-06T05:25:09.579960Z","shell.execute_reply":"2023-07-06T05:25:09.904481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Variable importances¶\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Get the feature importances from the model\nfeature_importances = catboost_model.feature_importances_\n\n# Get the names of the features used in the model\nfeature_names = train_df.drop(columns=['level_group', 'correct']).columns\n\n# Sort the feature importances in descending order\nsorted_indices = feature_importances.argsort()[::-1]\nsorted_feature_importances = feature_importances[sorted_indices]\nsorted_feature_names = feature_names[sorted_indices]\n\n# Plot the feature importances\nplt.figure(figsize=(10, 6))\nplt.bar(range(len(feature_importances)), sorted_feature_importances, tick_label=sorted_feature_names)\nplt.xticks(rotation=90)\nplt.xlabel('Features')\nplt.ylabel('Importance')\nplt.title('Variable Importances')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:09.906602Z","iopub.execute_input":"2023-07-06T05:25:09.906916Z","iopub.status.idle":"2023-07-06T05:25:10.183600Z","shell.execute_reply.started":"2023-07-06T05:25:09.906889Z","shell.execute_reply":"2023-07-06T05:25:10.182482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/sample_submission.csv')\n# ADD EXTRA COLUMNS\n#submission_df['session'] = submission_df.session_id.apply(lambda x: int(x.split('_')[0]) )\n#submission_df['q'] = submission_df.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:10.187044Z","iopub.execute_input":"2023-07-06T05:25:10.187449Z","iopub.status.idle":"2023-07-06T05:25:10.224168Z","shell.execute_reply.started":"2023-07-06T05:25:10.187420Z","shell.execute_reply":"2023-07-06T05:25:10.222476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder_310\ntry:\n    jo_wilder_310.make_env.__called__ = False\n    env.__called__ = False\n    type(env)._state = type(type(env)._state).__dict__['INIT']\nexcept:\n    pass\n\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()  ","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:10.226226Z","iopub.execute_input":"2023-07-06T05:25:10.226654Z","iopub.status.idle":"2023-07-06T05:25:10.253537Z","shell.execute_reply.started":"2023-07-06T05:25:10.226624Z","shell.execute_reply":"2023-07-06T05:25:10.252676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    test_df = test_df.drop(columns=['level_group']).values\n\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        predictions = gbtm.predict(test_df)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n      \n        sample_submission.loc[mask, 'correct'] = predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:10.257101Z","iopub.execute_input":"2023-07-06T05:25:10.257525Z","iopub.status.idle":"2023-07-06T05:25:10.719262Z","shell.execute_reply.started":"2023-07-06T05:25:10.257494Z","shell.execute_reply":"2023-07-06T05:25:10.718047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:10.720972Z","iopub.execute_input":"2023-07-06T05:25:10.721324Z","iopub.status.idle":"2023-07-06T05:25:10.743438Z","shell.execute_reply.started":"2023-07-06T05:25:10.721297Z","shell.execute_reply":"2023-07-06T05:25:10.742311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:25:10.744958Z","iopub.execute_input":"2023-07-06T05:25:10.745999Z","iopub.status.idle":"2023-07-06T05:25:11.051811Z","shell.execute_reply.started":"2023-07-06T05:25:10.745968Z","shell.execute_reply":"2023-07-06T05:25:11.050761Z"},"trusted":true},"execution_count":null,"outputs":[]}]}