{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The idea of this notebook is to analyse the data (perform EDA) to understand the usefulness of each feature. However the data is analysed for one certain level which could later on be used for other levels given the size of the data.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-29T04:47:23.526817Z","iopub.execute_input":"2023-05-29T04:47:23.529814Z","iopub.status.idle":"2023-05-29T04:47:24.934042Z","shell.execute_reply.started":"2023-05-29T04:47:23.529757Z","shell.execute_reply":"2023-05-29T04:47:24.932999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"markdown","source":"#### Since longer time is consumed in loading the data. The datatypes are changed to more optimized format to reduce the size.\n","metadata":{}},{"cell_type":"markdown","source":"Below are the datatypes of each feature:\n\n\n*     session_id      -int64  \n*     index           -int64  \n*     elapsed_time    -int64  \n*     event_name      -object \n*     name            -object \n*     level           -int64  \n*     page            -float64\n*     room_coor_x     -float64\n*     room_coor_y     -float64\n*     screen_coor_x   -float64\n*     screen_coor_y   -float64\n*     hover_duration  -float64\n*     text            -object \n*     fqid            -object \n*     room_fqid       -object \n*     text_fqid       -object \n*     fullscreen      -int64  \n*     hq              -int64  \n*     music           -int64  \n*     level_group     -object ","metadata":{}},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:48:04.616668Z","iopub.execute_input":"2023-05-29T04:48:04.617093Z","iopub.status.idle":"2023-05-29T04:48:04.624815Z","shell.execute_reply.started":"2023-05-29T04:48:04.617057Z","shell.execute_reply":"2023-05-29T04:48:04.623800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:48:30.000548Z","iopub.execute_input":"2023-05-29T04:48:30.001361Z","iopub.status.idle":"2023-05-29T04:50:46.199817Z","shell.execute_reply.started":"2023-05-29T04:48:30.001311Z","shell.execute_reply":"2023-05-29T04:50:46.198733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EXPLORATORY DATA ANALYSIS","metadata":{}},{"cell_type":"code","source":"#Printing the shape of the data:\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-28T15:38:59.823265Z","iopub.execute_input":"2023-05-28T15:38:59.823783Z","iopub.status.idle":"2023-05-28T15:38:59.836576Z","shell.execute_reply.started":"2023-05-28T15:38:59.823741Z","shell.execute_reply":"2023-05-28T15:38:59.835289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The data consists of 26296946 rows and 20 features","metadata":{}},{"cell_type":"code","source":"# Checking for Null values\nNull_Per = {}\nx = len(train)\nfor i in train.columns:    \n    Null_Per[i] = round((train[i].isnull().sum()/x)*100,1)\nprint(Null_Per)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:35:04.703450Z","iopub.execute_input":"2023-05-29T04:35:04.703978Z","iopub.status.idle":"2023-05-29T04:35:05.583358Z","shell.execute_reply.started":"2023-05-29T04:35:04.703908Z","shell.execute_reply":"2023-05-29T04:35:05.581139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It is observed from above that \"page\" has 97% null values and could be ignored for further analysis. \n#### The same could be applied for text and text_fqid\n#### However since hover_duration repesents only for hover events. This could used for analysis.","metadata":{}},{"cell_type":"code","source":"#Describing the data\ntrain.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:35:08.620704Z","iopub.execute_input":"2023-05-29T04:35:08.621245Z","iopub.status.idle":"2023-05-29T04:35:19.910842Z","shell.execute_reply.started":"2023-05-29T04:35:08.621198Z","shell.execute_reply":"2023-05-29T04:35:19.909577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Some basic understanding of some of the terms describing:\n#### Standard Deviation: The amount of variance from the mean. Low Std dev indicates that the values tend to be closed to the mean. While a high Std dev indicates that the values are spread out.\n#### The following observation could be made on each date:\n#### 1. elapsed_time: how much time has passed (in ms) between the start of the session and when the event was recorded\n#### 2. hover_duration - how long (in ms) the hover happened for (only for hover events)\n#### It is observed the standard deviation is extremly high for elapsed time and hover_duration as compared to other variables.","metadata":{}},{"cell_type":"code","source":"#Dropping session_id for further analysis\ndf = train.drop(['session_id'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:43:43.351316Z","iopub.execute_input":"2023-05-29T04:43:43.351956Z","iopub.status.idle":"2023-05-29T04:43:44.036847Z","shell.execute_reply.started":"2023-05-29T04:43:43.351899Z","shell.execute_reply":"2023-05-29T04:43:44.035253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dividing the data set into each group based on the level\n","metadata":{}},{"cell_type":"code","source":"df1 = df[df['level_group']=='0-4']\ndf2 = df[df['level_group']=='5-12']\ndf3 = df[df['level_group']=='13-22']","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:43:46.430329Z","iopub.execute_input":"2023-05-29T04:43:46.431604Z","iopub.status.idle":"2023-05-29T04:43:49.157103Z","shell.execute_reply.started":"2023-05-29T04:43:46.431548Z","shell.execute_reply":"2023-05-29T04:43:49.155717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Due to the vast dataset, the EDA is performed on level(0-1). The same could be used for rest of the levels for analysis.","metadata":{}},{"cell_type":"markdown","source":"## Univariate Analysis of Categorical Variables using Frequency Plots","metadata":{}},{"cell_type":"code","source":"#Segregating Catgorical Variabeles\nCatg = []\nfor i in df.columns:\n    if df[i].dtype=='category':\n        Catg.append(i)\nprint(Catg)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:43:50.411548Z","iopub.execute_input":"2023-05-29T04:43:50.412037Z","iopub.status.idle":"2023-05-29T04:43:50.421227Z","shell.execute_reply.started":"2023-05-29T04:43:50.411989Z","shell.execute_reply":"2023-05-29T04:43:50.420015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### A Bar Plot could be used for Categorical Data Analysis.","metadata":{}},{"cell_type":"code","source":"#event_name - the name of the event type\ndf1['event_name'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:43:53.929315Z","iopub.execute_input":"2023-05-29T04:43:53.930353Z","iopub.status.idle":"2023-05-29T04:43:54.369624Z","shell.execute_reply.started":"2023-05-29T04:43:53.930297Z","shell.execute_reply":"2023-05-29T04:43:54.368226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Every page should have a navigate click. Hence Navigate click could have highest counts","metadata":{}},{"cell_type":"code","source":"#name - the event name (e.g. identifies whether a notebook_click is is opening or closing the notebook)\ndf1['name'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:43:57.108667Z","iopub.execute_input":"2023-05-29T04:43:57.109640Z","iopub.status.idle":"2023-05-29T04:43:57.357223Z","shell.execute_reply.started":"2023-05-29T04:43:57.109587Z","shell.execute_reply":"2023-05-29T04:43:57.355550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It is observered that undefined has maximum counts and we could be assumed as one of the categories.","metadata":{}},{"cell_type":"code","source":"#text - the text the player sees during this event\nplt.figure(figsize=(40,5))\ndf1['text'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:44:00.430748Z","iopub.execute_input":"2023-05-29T04:44:00.431267Z","iopub.status.idle":"2023-05-29T04:44:18.550932Z","shell.execute_reply.started":"2023-05-29T04:44:00.431219Z","shell.execute_reply":"2023-05-29T04:44:18.548089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It is observered that 'undefined' has maximum counts.However for further analysis we could also assume 'undefined' as one of the unique values for further analysis and model training","metadata":{}},{"cell_type":"code","source":"#fqid - the fully qualified ID of the event\nplt.figure(figsize=(20,5))\ndf1['fqid'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:44:59.954405Z","iopub.execute_input":"2023-05-29T04:44:59.955046Z","iopub.status.idle":"2023-05-29T04:45:02.788948Z","shell.execute_reply.started":"2023-05-29T04:44:59.954991Z","shell.execute_reply":"2023-05-29T04:45:02.787630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### This could also be a useful variable for our model. Since the correctness of the question could also depend on the type of id. It could determine if groupconvo/ report/gramps and so on are involved in the answering the question","metadata":{}},{"cell_type":"code","source":"#room_fqid - the fully qualified ID of the room the event took place in\ndf1['room_fqid'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:06.913633Z","iopub.execute_input":"2023-05-29T04:45:06.914500Z","iopub.status.idle":"2023-05-29T04:45:07.362600Z","shell.execute_reply.started":"2023-05-29T04:45:06.914450Z","shell.execute_reply":"2023-05-29T04:45:07.360490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#text_fqid - the fully qualified ID of the\nplt.figure(figsize=(20,5))\ndf1['text_fqid'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:12.880672Z","iopub.execute_input":"2023-05-29T04:45:12.881204Z","iopub.status.idle":"2023-05-29T04:45:18.058798Z","shell.execute_reply.started":"2023-05-29T04:45:12.881154Z","shell.execute_reply":"2023-05-29T04:45:18.056455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 'text_fqid' could be lot more helpful since it combines above 2 variables.Hence we can ignore fqid and room_fquid individually.","metadata":{}},{"cell_type":"code","source":"#level - what level of the game the event occurred in (0 to 22)\n#We can also include level in Categorical type given the discrete type of values.\ndf1['level'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:20.866503Z","iopub.execute_input":"2023-05-29T04:45:20.867335Z","iopub.status.idle":"2023-05-29T04:45:21.101149Z","shell.execute_reply.started":"2023-05-29T04:45:20.867280Z","shell.execute_reply":"2023-05-29T04:45:21.099584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It is observed that level 3 has highest frequency compared to other levels. For our model since we would group the variables based on count, we would understad how the level would effect the correctness of the question.","metadata":{}},{"cell_type":"code","source":"df1['fullscreen'].value_counts()/len(df1)*100","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:24.616630Z","iopub.execute_input":"2023-05-29T04:45:24.617135Z","iopub.status.idle":"2023-05-29T04:45:24.652117Z","shell.execute_reply.started":"2023-05-29T04:45:24.617069Z","shell.execute_reply":"2023-05-29T04:45:24.650700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1['hq'].value_counts()/len(df1)*100","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:26.871274Z","iopub.execute_input":"2023-05-29T04:45:26.871777Z","iopub.status.idle":"2023-05-29T04:45:26.908101Z","shell.execute_reply.started":"2023-05-29T04:45:26.871730Z","shell.execute_reply":"2023-05-29T04:45:26.906825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1['music'].value_counts()/len(df1)*100","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:28.903518Z","iopub.execute_input":"2023-05-29T04:45:28.904048Z","iopub.status.idle":"2023-05-29T04:45:28.939140Z","shell.execute_reply.started":"2023-05-29T04:45:28.904001Z","shell.execute_reply":"2023-05-29T04:45:28.937880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### From the above it could be observed that most of the sessions have:\n#### 1. If 0 is considered as not fullscreen. 86% of the sessions are not on fullscreen,\n#### 2. 87% of data is not hq\n#### 3. 92% of sessions are with music on.\n### Since most of the data is baised to either one category, we could ignore these variables for our model","metadata":{}},{"cell_type":"markdown","source":"# Univariate Analysis of Continous Variables using Box Plots","metadata":{}},{"cell_type":"code","source":"#We can segregate continous variables as follow\nCont = []\nfor i in df1.columns:\n    if df1[i].dtype!='category':\n        Cont.append(i)\nprint(Cont)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:35.455894Z","iopub.execute_input":"2023-05-29T04:45:35.456800Z","iopub.status.idle":"2023-05-29T04:45:35.464983Z","shell.execute_reply.started":"2023-05-29T04:45:35.456745Z","shell.execute_reply":"2023-05-29T04:45:35.463665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 'index' can be ignored for now. 'page' which is  the page number of the event (only for notebook-related events), would also not be relevant for the anlysis given 97% of the data is null. 'level, 'fullscreen','hq' and 'music' have been discussed in categorical type. Also hover_duration which is  how long (in ms) the hover happened for (only for hover events) could be ignored for outliers analysis.","metadata":{}},{"cell_type":"code","source":"df1_Cont = df1.drop(['event_name', 'name', 'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group',\n                 'index','level','page','fullscreen','hq','music','hover_duration','level_group',],axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:39.439499Z","iopub.execute_input":"2023-05-29T04:45:39.440678Z","iopub.status.idle":"2023-05-29T04:45:39.495983Z","shell.execute_reply.started":"2023-05-29T04:45:39.440624Z","shell.execute_reply":"2023-05-29T04:45:39.494872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['elapsed_time', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y']\nfor i in cols:\n    plt.figure(figsize=(5,3))\n    sns.boxplot(df1_Cont[i])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T04:45:51.225582Z","iopub.execute_input":"2023-05-29T04:45:51.226421Z","iopub.status.idle":"2023-05-29T04:45:54.369616Z","shell.execute_reply.started":"2023-05-29T04:45:51.226355Z","shell.execute_reply":"2023-05-29T04:45:54.368109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Multivariate Analysis of Continous Variables using Scatter Plot","metadata":{}},{"cell_type":"markdown","source":"### The multivariate analysis is performed on Correctness for a question, that we could acquire from 'train_labels' data. Since the common feature among the train and labels data is 'session_id', it is important that we group by the variables per session_id and level_group of the train data set. We assign mean to continous values and nunique for Categorical Values. Here we are using only level0-4 for our analysis.","metadata":{}},{"cell_type":"code","source":"#Filering data of with level between 0 to 4\ndf_g = train[train['level_group']=='0-4']","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:00:46.753262Z","iopub.execute_input":"2023-05-29T05:00:46.753806Z","iopub.status.idle":"2023-05-29T05:00:47.365798Z","shell.execute_reply.started":"2023-05-29T05:00:46.753758Z","shell.execute_reply":"2023-05-29T05:00:47.364145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ContF =  ['elapsed_time', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']\ndfs = []\nfor c in df_g.columns:\n    if c in ContF:        \n        tmp = df_g.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)        \n    elif c in CatF:\n        tmp = df_g.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:25:25.929781Z","iopub.execute_input":"2023-05-29T05:25:25.930654Z","iopub.status.idle":"2023-05-29T05:25:34.635272Z","shell.execute_reply.started":"2023-05-29T05:25:25.930607Z","shell.execute_reply":"2023-05-29T05:25:34.634014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df1 = pd.concat(dfs,axis=1)\ndataset_df1 = dataset_df1.reset_index()\ndataset_df1 = dataset_df1.set_index('session_id')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:25:50.627010Z","iopub.execute_input":"2023-05-29T05:25:50.628048Z","iopub.status.idle":"2023-05-29T05:25:50.710416Z","shell.execute_reply.started":"2023-05-29T05:25:50.628000Z","shell.execute_reply":"2023-05-29T05:25:50.709245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Reading Label data","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\nlabels[['session_id', 'question']] = labels['session_id'].str.split('_', expand=True)\nlabels = labels.pivot(columns='question', index='session_id', values='correct')\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:25:57.669416Z","iopub.execute_input":"2023-05-29T05:25:57.670146Z","iopub.status.idle":"2023-05-29T05:26:00.692884Z","shell.execute_reply.started":"2023-05-29T05:25:57.670096Z","shell.execute_reply":"2023-05-29T05:26:00.691774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We will combine the scores of each level for Multivariate Analysis\nScores_l1 = pd.DataFrame(labels['q1']+labels['q2']+labels['q3']+labels['q4'],columns=['Scores0_1'])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:26:03.644708Z","iopub.execute_input":"2023-05-29T05:26:03.645595Z","iopub.status.idle":"2023-05-29T05:26:03.654817Z","shell.execute_reply.started":"2023-05-29T05:26:03.645546Z","shell.execute_reply":"2023-05-29T05:26:03.653532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# These scores from label_data is combined with Train data\ndf_g1 = pd.concat([dataset_df1.reset_index(),Scores_l1.reset_index()],axis=1,join='inner')","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:45:02.920489Z","iopub.execute_input":"2023-05-29T05:45:02.921366Z","iopub.status.idle":"2023-05-29T05:45:02.948984Z","shell.execute_reply.started":"2023-05-29T05:45:02.921316Z","shell.execute_reply":"2023-05-29T05:45:02.947926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Since here our y variable to perform Hypothesis Testing is continous. The following tests would be used \n### If x variable is Discrete: ANOVA\n### If x variable is continous: CORRELATION","metadata":{}},{"cell_type":"markdown","source":"# Hypothesis Testing for Y variable that is CONTINOUS and X variable that is CONTINOUS using CORRELATION","metadata":{}},{"cell_type":"markdown","source":"### A heatmap could be used to understand the correlation of the scores with rest of the independent continous variables.\n### Since heatmap determs the correlation, it would be more useful for continous data. The Categorical data is removed.","metadata":{}},{"cell_type":"code","source":"CatF = ['session_id','level_group', 'session_id_nunique','event_name_nunique', 'name_nunique', 'level_nunique', 'fqid_nunique', 'room_fqid_nunique', 'level_group_nunique']","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:27:29.522786Z","iopub.execute_input":"2023-05-29T05:27:29.524051Z","iopub.status.idle":"2023-05-29T05:27:29.529281Z","shell.execute_reply.started":"2023-05-29T05:27:29.524001Z","shell.execute_reply":"2023-05-29T05:27:29.528200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Filtering all the variables of Continous type\ndf_g1.drop('session_id',axis=1,inplace=True)\ndf_g1_Cont = df_g1.copy()\nfor i in df_g1_Cont.columns:\n    if i in CatF:\n        df_g1_Cont.drop(i,axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:28:26.037165Z","iopub.execute_input":"2023-05-29T05:28:26.037900Z","iopub.status.idle":"2023-05-29T05:28:26.055251Z","shell.execute_reply.started":"2023-05-29T05:28:26.037860Z","shell.execute_reply":"2023-05-29T05:28:26.053889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creating a heatmap\nplt.figure(figsize=(10,3))\nsns.heatmap(df_g1_Cont.corr(),annot=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:05.885105Z","iopub.execute_input":"2023-05-29T05:30:05.885558Z","iopub.status.idle":"2023-05-29T05:30:06.597627Z","shell.execute_reply.started":"2023-05-29T05:30:05.885519Z","shell.execute_reply":"2023-05-29T05:30:06.596534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The correlation of Scores with rest of the features seem to be less.\n####  However we can as well visualse the data using scatter plot","metadata":{}},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['elapsed_time_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"elapsed_time_mean\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:26.526886Z","iopub.execute_input":"2023-05-29T05:30:26.527322Z","iopub.status.idle":"2023-05-29T05:30:26.816154Z","shell.execute_reply.started":"2023-05-29T05:30:26.527284Z","shell.execute_reply":"2023-05-29T05:30:26.815001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['room_coor_x_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"room_coor_x_mean\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:30.018459Z","iopub.execute_input":"2023-05-29T05:30:30.019484Z","iopub.status.idle":"2023-05-29T05:30:30.275717Z","shell.execute_reply.started":"2023-05-29T05:30:30.019424Z","shell.execute_reply":"2023-05-29T05:30:30.274661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['room_coor_y_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"room_coor_y_mean\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:32.434473Z","iopub.execute_input":"2023-05-29T05:30:32.434899Z","iopub.status.idle":"2023-05-29T05:30:32.693860Z","shell.execute_reply.started":"2023-05-29T05:30:32.434861Z","shell.execute_reply":"2023-05-29T05:30:32.692813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['screen_coor_x_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"screen_coor_x_mean\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:34.832634Z","iopub.execute_input":"2023-05-29T05:30:34.833553Z","iopub.status.idle":"2023-05-29T05:30:35.082892Z","shell.execute_reply.started":"2023-05-29T05:30:34.833302Z","shell.execute_reply":"2023-05-29T05:30:35.081836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['screen_coor_y_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"screen_coor_y_mean\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:37.707162Z","iopub.execute_input":"2023-05-29T05:30:37.707880Z","iopub.status.idle":"2023-05-29T05:30:37.969552Z","shell.execute_reply.started":"2023-05-29T05:30:37.707839Z","shell.execute_reply":"2023-05-29T05:30:37.968533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['screen_coor_y_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"screen_coor_y_mean\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:40.902261Z","iopub.execute_input":"2023-05-29T05:30:40.902703Z","iopub.status.idle":"2023-05-29T05:30:41.166849Z","shell.execute_reply.started":"2023-05-29T05:30:40.902663Z","shell.execute_reply":"2023-05-29T05:30:41.165447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(df_g1_Cont['Scores0_1'],df_g1_Cont['hover_duration_mean'])\nplt.xlabel(\"Scores0_1\")\nplt.ylabel(\"hover_duration_mean\")","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:43.431253Z","iopub.execute_input":"2023-05-29T05:30:43.431707Z","iopub.status.idle":"2023-05-29T05:30:43.699944Z","shell.execute_reply.started":"2023-05-29T05:30:43.431668Z","shell.execute_reply":"2023-05-29T05:30:43.698514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We can once again confirm that the scores are equally distributed for change for range of values for each features.","metadata":{}},{"cell_type":"markdown","source":"# Hypothesis Testing for Y variable that is CONTINOUS and X variable that is DISCRETE using ANOVA","metadata":{}},{"cell_type":"markdown","source":"### In ANOVA the following are to be tested:\n##### A null hypothesis (H0): This is when there is no difference between the groups or means. \n##### An alternative hypothesis (H1): When it is theorized that there is a difference between groups and means.\n##### If the F statistic is higher than the critical value (the value of F that corresponds with your alpha value, usually 0.05), then the difference among groups is deemed statistically significant. We fail to accept the Null Hypothesis and prove that the means of different groups are different. A significance level (denoted as α or alpha) of 0.05 works well. A significance level of 0.05 indicates a 5% risk of concluding that a difference exists when there is no actual difference.","metadata":{}},{"cell_type":"code","source":"from statsmodels.formula.api import ols      # For n-way ANOVA\nfrom statsmodels.stats.anova import _get_covariance,anova_lm # For n-way ANOVA","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:30:50.588707Z","iopub.execute_input":"2023-05-29T05:30:50.589147Z","iopub.status.idle":"2023-05-29T05:30:51.105252Z","shell.execute_reply.started":"2023-05-29T05:30:50.589105Z","shell.execute_reply":"2023-05-29T05:30:51.104214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ContF =  ['elapsed_time_mean', 'room_coor_x_mean', 'room_coor_y_mean', 'screen_coor_x_mean', 'screen_coor_y_mean', 'hover_duration_mean']\n#Filtering all the variables of Continous type\ndf_g1_Cat = df_g1.copy()\nfor i in df_g1_Cont.columns:\n    if i in ContF:\n        df_g1_Cat.drop(i,axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:43:35.604076Z","iopub.execute_input":"2023-05-29T05:43:35.605013Z","iopub.status.idle":"2023-05-29T05:43:35.624546Z","shell.execute_reply.started":"2023-05-29T05:43:35.604957Z","shell.execute_reply":"2023-05-29T05:43:35.623526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"formula_Name = 'Scores0_1 ~ C(event_name_nunique)+ C(name_nunique)'\nAnova_Name = ols(formula_Name, df_g1).fit()\naov_Name = anova_lm(Anova_Name)\nprint(aov_Name)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:46:29.479254Z","iopub.execute_input":"2023-05-29T05:46:29.480707Z","iopub.status.idle":"2023-05-29T05:46:29.910414Z","shell.execute_reply.started":"2023-05-29T05:46:29.480637Z","shell.execute_reply":"2023-05-29T05:46:29.909088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"formula_fqid = 'Scores0_1 ~ C(room_fqid_nunique)'\nAnova_fqid = ols(formula_fqid, df_g1).fit()\naov_fqid = anova_lm(Anova_fqid)\nprint(aov_fqid)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T05:46:37.537966Z","iopub.execute_input":"2023-05-29T05:46:37.538377Z","iopub.status.idle":"2023-05-29T05:46:37.750480Z","shell.execute_reply.started":"2023-05-29T05:46:37.538339Z","shell.execute_reply":"2023-05-29T05:46:37.748291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lower the p value, higher the F value and hence greater the variation between the means. Which indicates each event has a no significant impact on the scores/performance of the user.","metadata":{}},{"cell_type":"markdown","source":"# SUMMARY","metadata":{}},{"cell_type":"markdown","source":"#### 1. The data consists of several features which have more than 70% of null values and these features could be dropped for model training\n#### 2. With ANOVA test, it is confirmed that some of the variables have not much impact on scoring of the user. However since the test has been conducted for level 0-4, it is best to retain the features since other levels are also to be tested.","metadata":{}}]}