{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-06T18:42:01.757265Z","iopub.execute_input":"2023-06-06T18:42:01.757972Z","iopub.status.idle":"2023-06-06T18:42:01.783298Z","shell.execute_reply.started":"2023-06-06T18:42:01.757931Z","shell.execute_reply":"2023-06-06T18:42:01.782202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading the dataset**\n\nSince the dataset is huge, some people may face memory errors while reading the dataset from the csv. To avoid this, we will try to optimize the memory used by Pandas to load and store the dataset.\n\nWhen Pandas loads a dataset, by default, it automatically detects the data types of the different columns. Irresepective of the maximum value that is stored in these columns, Pandas assigns int64 for numerical columns, float64 for float columns, object dtype for string columns etc.\n\nWe may be able to reduce the size of these columns in memory by downcasting numerical columns to smaller types (like int8, int32, float32 etc.), if their maximum values don't need the larger types for storage, (like int64, float64 etc.).\n\nSimilarly, Pandas automatically detects string columns as object datatype. To reduce memory usage of string columns which store categorical data, we specify their datatype as category.\n\nMany of the columns in this dataset can be downcast to smaller types.\n\nWe will provide a dict of dtypes for columns to pandas while reading the dataset.\n\nRef: https://www.kaggle.com/code/gusthema/student-performance-w-tensorflow-decision-forests","metadata":{}},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndf_train_text = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\ndf_test_text = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv', dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:42:01.784904Z","iopub.execute_input":"2023-06-06T18:42:01.785539Z","iopub.status.idle":"2023-06-06T18:43:45.136095Z","shell.execute_reply.started":"2023-06-06T18:42:01.785499Z","shell.execute_reply":"2023-06-06T18:43:45.134761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading the labeled dataset**","metadata":{}},{"cell_type":"code","source":"df_train_labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:45.137793Z","iopub.execute_input":"2023-06-06T18:43:45.138282Z","iopub.status.idle":"2023-06-06T18:43:45.501527Z","shell.execute_reply.started":"2023-06-06T18:43:45.138244Z","shell.execute_reply":"2023-06-06T18:43:45.500256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels[['Session_ID_labels', 'question']] =  df_train_labels['session_id'].str.split('_',expand=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:45.505341Z","iopub.execute_input":"2023-06-06T18:43:45.505701Z","iopub.status.idle":"2023-06-06T18:43:47.075668Z","shell.execute_reply.started":"2023-06-06T18:43:45.505666Z","shell.execute_reply":"2023-06-06T18:43:47.074371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the dataset, i.e., train and test","metadata":{}},{"cell_type":"code","source":"class exploreData():\n    \n    def __init__(self, train, test):\n        self.train = train\n        self.test = test\n       \n    def trainHead(self):\n        return self.train.head()\n    \n    def testHead(self):\n        return self.test.head()\n    \n    def trainDataset_description(self):\n        return self.train.describe()\n    \n    def testDataset_description(self):\n        return self.test.describe()\n    \n    def column_info(self):\n        \n        print('''We will be underdtading the basic structure of the dataset that we have. This includes general information like:\n        \n            * Column data types\n            \n            ''')\n        \n        print('Columns for the training dataset looks like:')\n        train_data = self.train.dtypes\n        \n        print('\\n')\n        \n        print('Columns for the training dataset looks like:')\n        test_data = self.test.dtypes\n        \n        return train_data, test_data\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:47.077537Z","iopub.execute_input":"2023-06-06T18:43:47.077926Z","iopub.status.idle":"2023-06-06T18:43:47.088137Z","shell.execute_reply.started":"2023-06-06T18:43:47.077889Z","shell.execute_reply":"2023-06-06T18:43:47.087077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    print(\"Before going any further, I want to see how the training and testing data looks like:\")\n    print('\\n')\n    exploredata = exploreData(df_train_text,df_test_text)\n    print(\"Let's see how the train data looks like:\")\n    print('\\n')\n    exploredata.trainHead()\n    print(exploredata.trainDataset_description())\n    print('\\n')\n    print(\"Let's see how our test data looks like:\")\n    exploredata.testHead()\n    print(exploredata.testDataset_description)\n    print('\\n')\n    print(exploredata.column_info())\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:47.089453Z","iopub.execute_input":"2023-06-06T18:43:47.090579Z","iopub.status.idle":"2023-06-06T18:43:57.780805Z","shell.execute_reply.started":"2023-06-06T18:43:47.090535Z","shell.execute_reply":"2023-06-06T18:43:57.779507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks like we have a lot of unstructured data in the form of \"category\" and \"object\" data types. Columns which are unstructured in nature are:\n\n1. text              category\n2. fqid              category\n3. room_fqid         category\n4. text_fqid         category\n5. fullscreen        category\n6. hq                category\n7. music             category\n8. level_group       category\n9. event_name        category\n10. name              category","metadata":{}},{"cell_type":"code","source":"class trainDataPreparation:\n    \n    def __init__(self):\n        pass\n\n    def merge(self):\n\n        global df_train_final\n        df_train_final = pd.concat([df_train_text, df_train_labels], axis = 1, join = 'inner')\n        return df_train_final\n    \n    def checkNA(self):\n        \n        self.merge()\n        \n        return df_train_final.isna().sum()\n    \n    def numberOfRows(self):\n        \n        number_of_rows = df_train_final.shape[0]\n        print('Total number of rows in the dataset:',number_of_rows)\n        \n        \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:57.782599Z","iopub.execute_input":"2023-06-06T18:43:57.782977Z","iopub.status.idle":"2023-06-06T18:43:57.791252Z","shell.execute_reply.started":"2023-06-06T18:43:57.782940Z","shell.execute_reply":"2023-06-06T18:43:57.790067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Let's initiate the above module and the 'merge' method to get the dataset that we want for our exploratory data analysis:\")\ntrainData = trainDataPreparation()\ntrainData.merge()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:57.793020Z","iopub.execute_input":"2023-06-06T18:43:57.793378Z","iopub.status.idle":"2023-06-06T18:43:58.085602Z","shell.execute_reply.started":"2023-06-06T18:43:57.793344Z","shell.execute_reply":"2023-06-06T18:43:58.084601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above class is a helper module which will be helpful for us in future to merge the text and their corresponding labels. It will also help us in letting us know the following details:\n\n1. NA values in the dataset\n2. number of rows in the dataset","metadata":{}},{"cell_type":"markdown","source":"**According to the data, not all of the events have fully qualified ID. \nBut since this project is about the events the children are enrolled in, I first need to see how many events are there.**\n\nTo explore the data regarding events, we need to consider the following features:\n\n1. event_name\n2. fqid\n3. room_fqid\n4. session_id\n5. question\n\nAlso we will be assessing various factors as to what happens if the moderator is asking a question to the player. We will try to understand the mentality of the student. Is he/she on later stages of the questions, on what level i=does the moderator needs to intervene and is the player hovering or clicking\n\nThen finally, after understanding the player mentality, now we will understand a couple of basic things:\n\nMonitor the student performance based on the number of pages in the games?\n\n1. Level group/ # of pages\n2. If hover time is less, then number of pages is less?","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nclass level_group_studentActivity:\n    \n    def __init__(self):\n        self.df_train = df_train_text\n        \n    def levelForAges(self):\n        level_and_ages = pd.DataFrame({\n                \"level\": self.df_train['level'],\n                \"page\": self.df_train['page']\n            })\n        \n        level_and_ages_final  = level_and_ages.dropna()\n        print(level_and_ages_final.groupby('level').first())\n        print('\\n')\n        return level_and_ages_final.groupby('level').first().plot()\n\nclass event(level_group_studentActivity):\n    \n    def __init__(self):\n        self.df_train = df_train_final\n        \n    def eventName(self):\n        return self.df_train['event_name'].value_counts()\n    \n    def eventName_ID(self):\n        return pd.pivot_table(data=self.df_train,index=['event_name'])\n    \n    def event_activity_fqid(self):\n        \n        global df_hovering\n        \n        df_activity_info = self.df_train[['event_name', 'fqid', 'room_fqid', 'correct', 'question','fullscreen']]\n        max_correct = max(df_activity_info['correct'])\n        ## total number of questions\n        print('\\n')\n        print('Total number of questions per event:')\n        print(df_activity_info['question'].value_counts())\n        print(\"\\n\")\n        #df_activity_info.groupby('question').size().plot(kind='pie', autopct='%.2f')\n        print(pd.pivot_table(data = df_activity_info,index = 'event_name'))\n        print('\\n')\n        print(\"Activity based on hovering of mouse:\")\n        print('\\n')\n        df_hovering = df_activity_info.loc[df_activity_info['correct'] == max_correct]\n        df_hovering[['event_in_the_game','event_name_click']] = df_hovering['event_name'].str.split('_',expand=True)\n        print(df_hovering['event_name_click'].value_counts())\n        df_hovering[['room_fquid_0','room_fquid_1','room_fquid_2']] = df_hovering['room_fqid'].str.split('.',expand=True)\n        \n    def event_activity_room_fqid(self):\n        print(df_hovering['room_fquid_2'].value_counts())    \n        return df_hovering.groupby('room_fquid_2').size().plot(kind='bar')\n        \n    def event_inTheGame_max(self):\n        print(df_hovering['event_in_the_game'].value_counts())\n        return df_hovering.groupby('event_in_the_game').size().plot(kind='bar')\n    \n    def navigateRoomfqid(self):\n        return df_hovering.groupby('question').size().plot(kind='bar')\n    \n    def fullscreen_mode(self):\n        \n        return df_hovering['fullscreen'].value_counts().plot()\n    \nclass textEDA:\n    \n    def __init__(self):\n        self.df_train = df_train_final\n        \n    def prepareQuestionMarkdata(self):\n        global df_questionMark\n        self.df_train['QuestionMark_present'] = self.df_train['text'].str.endswith('?')\n        df_questionMark=self.df_train.loc[self.df_train['QuestionMark_present'] == True]\n        df_questionMark[['event_in_the_game','edf_questionMark']] = df_questionMark['event_name'].str.split('_',expand=True)\n        df_questionMark[['room_fquid_0','room_fquid_1','room_fquid_2']] = df_questionMark['room_fqid'].str.split('.',expand=True)\n        return df_questionMark\n    \n    def timeSpentinRoom(self):\n        plt.figure(figsize=(12,5))\n        print('\\n')\n        print(df_questionMark['room_fquid_2'].value_counts())\n        print('\\n')\n        index = list(df_questionMark['room_fquid_2'].value_counts().index)\n        values = list(df_questionMark['room_fquid_2'].value_counts().values)\n        plt.title(\"Player spent most time when a promt was given to them as a question\")\n        \n        return plt.plot(index,values)\n    \n    def whichLevel(self):\n        plt.figure(figsize=(12,5))\n        print('\\n')\n        print(df_questionMark['level'].value_counts())\n        print('\\n')\n        index = list(df_questionMark['level'].value_counts().index)\n        values = list(df_questionMark['level'].value_counts().values)\n        plt.title(\"On what level was the player in when the moderator asked them a question\")\n        \n        return plt.plot(index,values)\n     \n    def difficultQuestion(self):\n     \n        plt.figure(figsize=(12,5))\n        print('\\n')\n        print(df_questionMark['question'].value_counts())\n        print('\\n')\n        index = list(df_questionMark['question'].value_counts().index)\n        values = list(df_questionMark['question'].value_counts().values)\n        plt.title(\"On what level was the player in when the moderator asked them a question\")\n        \n        return plt.plot(index,values)\n    \n    def what_eventIntheGame(self):\n     \n        plt.figure(figsize=(12,5))\n        print('\\n')\n        print(df_questionMark['event_in_the_game'].value_counts())\n        print('\\n')\n        index = list(df_questionMark['event_in_the_game'].value_counts().index)\n        values = list(df_questionMark['event_in_the_game'].value_counts().values)\n        plt.title(\"What is the event in the Game\")\n        \n        return plt.plot(index,values)\n    \n    def what_playerIsDoing(self):\n        plt.figure(figsize=(12,5))\n        print('\\n')\n        print(df_questionMark['edf_questionMark'].value_counts())\n        print('\\n')\n        index = list(df_questionMark['edf_questionMark'].value_counts().index)\n        values = list(df_questionMark['edf_questionMark'].value_counts().values)\n        plt.title(\"Is the person clicking or doing something else when question is being asked?\")\n        \n        return plt.plot(index,values)\n    \n    \n    def howMuchTimeelapsed(self):\n        df_questionMark['elapsed_time'] = (df_questionMark['elapsed_time']/(1000*60))%60\n        print('Elapsed time column converted into minutes')\n        print('\\n')\n        max_timeelapsed = max(df_questionMark['elapsed_time'])\n        print('\\n')\n        df_questionMark_maxTime = df_questionMark.loc[df_questionMark['elapsed_time'] == max_timeelapsed]\n        \n        print(\"Details about the situation where maximum time elapsed:\")\n        print('\\n')\n        print(\"Question where maximum time was spent:\",df_questionMark_maxTime['question'].item())\n        print(\"What was the event in the game:\",df_questionMark_maxTime['event_in_the_game'].item())\n        print(\"Time spent (in minutes):\", df_questionMark_maxTime['elapsed_time'].item(),\"minutes\")\n        #print(\"Session ID where student spent time and the moderator had to ask a question:\",df_questionMark_maxTime['session_id'].item())\n        print(\"In what location the player was in the game:\", df_questionMark_maxTime['room_fquid_2'].item())\n        print(\"In what level player was in:\",df_questionMark_maxTime['level'].item())\n        print('\\n')\n        \n        return df_questionMark_maxTime\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:58.087560Z","iopub.execute_input":"2023-06-06T18:43:58.088308Z","iopub.status.idle":"2023-06-06T18:43:58.870861Z","shell.execute_reply.started":"2023-06-06T18:43:58.088266Z","shell.execute_reply":"2023-06-06T18:43:58.869817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_text['hover_duration'] = df_train_text['hover_duration']/(1000*60)%60\ndf_train_text.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:58.872412Z","iopub.execute_input":"2023-06-06T18:43:58.873151Z","iopub.status.idle":"2023-06-06T18:43:59.489989Z","shell.execute_reply.started":"2023-06-06T18:43:58.873101Z","shell.execute_reply":"2023-06-06T18:43:59.488621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"We will now explore the data:\")\nprint('\\n')\nlevelsofGame = level_group_studentActivity()\nlevelsofGame.levelForAges()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:43:59.491875Z","iopub.execute_input":"2023-06-06T18:43:59.492268Z","iopub.status.idle":"2023-06-06T18:44:00.296931Z","shell.execute_reply.started":"2023-06-06T18:43:59.492232Z","shell.execute_reply":"2023-06-06T18:44:00.295911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from the above information and graph we can see that the number of pages in educational games and learning is increasing linearly in a positive trend. This means that as the level increases, the number of pages that the student needs to encounter are also increasing.\n\nwe can assume from the above information that the learning curve increases steeply as the level increases","metadata":{}},{"cell_type":"code","source":"print(\"Let's explore all the events present in the data:\")\nprint(\"\\n\")\nEvents = event()\nprint(\"Events present in the game are:\")\nprint(\"\\n\")\nprint(Events.eventName())\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:00.298374Z","iopub.execute_input":"2023-06-06T18:44:00.298985Z","iopub.status.idle":"2023-06-06T18:44:00.373480Z","shell.execute_reply.started":"2023-06-06T18:44:00.298938Z","shell.execute_reply":"2023-06-06T18:44:00.372231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So the maximum number of events in the game are coming from the clicking of buttons/toggle bar while navigating while the student is saving his work through checkpoint very less. this could either mean that the student is able to finish a lot learning/exercise in a single sitting or he is loosing his/her concentration quickly and exiting the game","metadata":{}},{"cell_type":"code","source":"Events.eventName_ID()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:00.379939Z","iopub.execute_input":"2023-06-06T18:44:00.380374Z","iopub.status.idle":"2023-06-06T18:44:00.552653Z","shell.execute_reply.started":"2023-06-06T18:44:00.380335Z","shell.execute_reply":"2023-06-06T18:44:00.551383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From our assumption above, we can see that there is not much elapsed time when the student is at checkpoint and the student is constantly either hovering over the map section or clicking on the map. Hence, the student is getting stucked and is trying to see if he/she is heading towards the right direction or not.","metadata":{}},{"cell_type":"code","source":"Events.event_activity_fqid()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:00.554313Z","iopub.execute_input":"2023-06-06T18:44:00.554678Z","iopub.status.idle":"2023-06-06T18:44:01.490892Z","shell.execute_reply.started":"2023-06-06T18:44:00.554641Z","shell.execute_reply":"2023-06-06T18:44:01.489527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Following are the observation from the above sequence:\n\n1. There are 18 questions in total and every question has equal number of question.\n2. Even if we consider our assumption, a student in educational leaning is spending more time in clicking as compare to hovering which can be considered idling in some cases and not paying attention. hence, we can say that even if the student is finding the activities difficult, the interactivity in these games are making the students engaged in the activity.\n3. We also notice that most number of correct answers are coming from clicking on an object.","metadata":{}},{"cell_type":"code","source":"Events.event_activity_room_fqid()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:01.492394Z","iopub.execute_input":"2023-06-06T18:44:01.492760Z","iopub.status.idle":"2023-06-06T18:44:01.824492Z","shell.execute_reply.started":"2023-06-06T18:44:01.492723Z","shell.execute_reply":"2023-06-06T18:44:01.823433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Events.event_inTheGame_max()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:01.826327Z","iopub.execute_input":"2023-06-06T18:44:01.827059Z","iopub.status.idle":"2023-06-06T18:44:02.117413Z","shell.execute_reply.started":"2023-06-06T18:44:01.827017Z","shell.execute_reply":"2023-06-06T18:44:02.116289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Events.fullscreen_mode()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:02.119241Z","iopub.execute_input":"2023-06-06T18:44:02.120039Z","iopub.status.idle":"2023-06-06T18:44:02.280539Z","shell.execute_reply.started":"2023-06-06T18:44:02.119993Z","shell.execute_reply":"2023-06-06T18:44:02.279411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Events.navigateRoomfqid()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:02.282030Z","iopub.execute_input":"2023-06-06T18:44:02.282652Z","iopub.status.idle":"2023-06-06T18:44:02.591514Z","shell.execute_reply.started":"2023-06-06T18:44:02.282609Z","shell.execute_reply":"2023-06-06T18:44:02.590227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All the above graphs and information are extracted where there are maximum number of correct answers. Surprisingly, it was seen that even though the number of pages increases and the student is spending more time on checkpoint as well, but they are also answering correct questions when reached at later stages of the question.\n\nAlso, in most of the correct options, the student was not in Fullscreen mode.\n\nAnd as previously mentioned, a student in the game spends a lot time navigating. Hence, we can say that in order for a student to perform good, the student is navigating no the screen where he/she is on to make an informed decision.\n\nThe room where most of the coorect answers were found were in:\n\n1. Frontdesk\n2. Entry\n3. Center\n\n","metadata":{}},{"cell_type":"code","source":"print(\"Let us understand the unstructured part of the data:\")\nprint('\\n')\nunStructuredData = textEDA()\na = unStructuredData.prepareQuestionMarkdata()\nprint(\"Total number of instances where the moderator needed to ask question from the student:\", a.shape[0])\nunStructuredData.howMuchTimeelapsed()\nunStructuredData.difficultQuestion()\nunStructuredData.timeSpentinRoom()\nunStructuredData.whichLevel()\nunStructuredData.what_eventIntheGame()\nunStructuredData.what_playerIsDoing()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:02.593397Z","iopub.execute_input":"2023-06-06T18:44:02.593747Z","iopub.status.idle":"2023-06-06T18:44:04.105286Z","shell.execute_reply.started":"2023-06-06T18:44:02.593713Z","shell.execute_reply":"2023-06-06T18:44:04.104042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Text Analysis for all the comments**","metadata":{}},{"cell_type":"markdown","source":"Since we have a lot information in unstructure format, I wanted to use graph theory to further analyze and understand the data. Since we are dealing with hige amount of data, we wanted to select features based on following parameters:\n\n1. Where questions were asked \n2. When the event in the game was based on \"person_click\"\n3. When level group was on 5-12\n4. When fqid was a worker\n5. When text_gqid =  tunic.drycleaner.frontdesk.worker.hub","metadata":{}},{"cell_type":"markdown","source":"Following commands were executed to prepare the data:\n\n1. df_train_text['Source_questionMark'] = df_train_text['text'].str.endswith('?')\n2. df_questionMark_network=df_train_text.loc[df_train_text['Source_questionMark'] == True]\n3. df_questionMark_network_final = df_questionMark_network.loc[df_questionMark_network['event_name'] == 'person_click']\n4. df_questionMark_network_final = df_questionMark_network.loc[df_questionMark_network['level_group'] == '5-12']\n5. df_questionMark_network_final = df_questionMark_network.loc[df_questionMark_network['fqid'] == 'worker']\n6. df_questionMark_network_final = df_questionMark_network.loc[df_questionMark_network['text_fqid'] == 'tunic.drycleaner.frontdesk.worker.hub']\n\ndf_questionMark_network_final.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-02T01:08:14.906032Z","iopub.execute_input":"2023-06-02T01:08:14.907010Z","iopub.status.idle":"2023-06-02T01:08:18.463308Z","shell.execute_reply.started":"2023-06-02T01:08:14.906954Z","shell.execute_reply":"2023-06-02T01:08:18.462033Z"}}},{"cell_type":"code","source":"def prepareData(df):\n    \n    \n    df[['event_id','event_name_activity']] = df['event_name'].str.split('_',expand = True) \n\n    df['Source_questionMark'] = df['text'].str.endswith('?')\n    df=df.loc[df['Source_questionMark'] == True]\n    df = df.loc[df['event_name'] == 'person_click']\n    df = df.loc[df['level_group'] == '5-12']\n    df = df.loc[df['fqid'] == 'worker']\n    df = df.loc[df['text_fqid'] == 'tunic.drycleaner.frontdesk.worker.hub']\n    \n    df_network = pd.DataFrame({\n        \"Source\": df['event_name_activity'],\n        'Target': df['text'],\n        'Weight': df['fullscreen']\n    })\n    return df_network\n    \n    \n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:04.106997Z","iopub.execute_input":"2023-06-06T18:44:04.107386Z","iopub.status.idle":"2023-06-06T18:44:04.116911Z","shell.execute_reply.started":"2023-06-06T18:44:04.107348Z","shell.execute_reply":"2023-06-06T18:44:04.115514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import networkx as nx\n\nclass createGraph:\n    \n    def __init__(self):\n        self.graphData = prepareData(df_train_text)\n    \n    def createGraphfromData(self):\n        # Create an empty directed graph\n        G = nx.Graph()\n        # Add edges to the graph\n        for index, row in self.graphData.iterrows():\n            source = row['Source']\n            target = row['Target']\n            weight = row['Weight']\n            G.add_edge(source, target, weight=weight)\n        print(\"Our graph looks like this:\")\n       \n        nx.draw(G, with_labels=True)\n        # Visualize the graph\n        return G\n        \n        \n\nclass networkTextAnalysis(createGraph):\n    \n    \n    def basicInfo_onGraph(self):\n        \n        global graph\n        \n        '''\n        nx.info() will tell us the following things:\n            1. number of nodes\n            2. number of edges\n            3. average degree\n            \n            what is average degree: Average degree is the average number of connections of each node in your network\n            '''\n        graph = super().createGraphfromData() \n        \n        return nx.info(graph)\n    \n    # Calculate the degree of each node\n    def degs(self):\n        degree = nx.degree(graph)\n        print('degree:',degree)\n        print('\\n')\n        \n        # Calculate the degree distribution\n        degree_sequence = [d for n, d in nx.degree(graph)]\n        degree_counts = nx.degree_histogram(graph)\n        print('degree sequence:',degree_sequence)\n        print('degree counts:',degree_counts)\n        \n    \n    # Calculate the number of connected components in the graph\n    def graphComponents(self):\n        \n        components = nx.connected_components(graph)\n        print('components:',components)\n        print('\\n')\n        return components\n    \n    # Calculate the average shortest path length between all pairs of nodes\n    def shortestPath(self):\n        shortest_path_length = nx.average_shortest_path_length(graph)\n        # Find all shortest paths\n        shortest_paths = dict(nx.all_pairs_shortest_path(graph))\n\n        # Print all shortest paths\n        for source, paths in shortest_paths.items():\n            for target, path in paths.items():\n                print(f\"Shortest path from {source} to {target}: {path}\")\n                \n        print('Average distance shortest path: ',shortest_path_length)\n        print('\\n')\n        return shortest_path_length\n    \n    # Calculate the clustering coefficient of each node\n    def clustering_coeff(self):\n        clustering_coefficient = nx.clustering(graph)\n        print('clustering coeff: ',clustering_coefficient)\n        print('\\n')\n        return clustering_coefficient\n    \n    # Calculate the betweenness centrality of each node\n    def bet_centrality(self):\n        betweenness_centrality = nx.betweenness_centrality(graph)\n        print('betweenness centrality: ',betweenness_centrality)\n        print('\\n')\n        nx.draw(graph, with_labels=True, node_color=list(betweenness_centrality.values()))\n        plt.show()\n        return betweenness_centrality\n    \n    # Calculate the closeness centrality of each node\n    def close_centrality(self):\n        closeness_centrality = nx.closeness_centrality(graph)\n        print('closenness centrality: ',closeness_centrality.items())\n        print('\\n')\n        nx.draw(graph, with_labels=True, node_color=list(closeness_centrality.values()))\n        plt.show()\n        return closeness_centrality\n    \n    # Create a dictionary of values and their corresponding graphs\n    def plot_values(self):\n        degree = self.degs()\n        components = self.graphComponents()\n        shortest_path_length = self.shortestPath()\n        clustering_coefficient = self.clustering_coeff()\n        betweenness_centrality = self.bet_centrality()\n        closeness_centrality = self.close_centrality\n        return degree, components, shortest_path_length, clustering_coefficient, betweenness_centrality, closeness_centrality \n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:04.118806Z","iopub.execute_input":"2023-06-06T18:44:04.119223Z","iopub.status.idle":"2023-06-06T18:44:04.267633Z","shell.execute_reply.started":"2023-06-06T18:44:04.119183Z","shell.execute_reply":"2023-06-06T18:44:04.266448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"textAnalysis = networkTextAnalysis()\nprint(\"The graph that has been formed on the data we have in hand has the following properties:\")\nprint('\\n')\ntextAnalysis.basicInfo_onGraph()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:04.269047Z","iopub.execute_input":"2023-06-06T18:44:04.270065Z","iopub.status.idle":"2023-06-06T18:44:47.215052Z","shell.execute_reply.started":"2023-06-06T18:44:04.270023Z","shell.execute_reply":"2023-06-06T18:44:47.213873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the above \"Directed weighted\" graph, we can say that:\n\nThe click action performed by the student indicates that the moderator is trying to ask some sort of question when they are stuck somewhere or the student is trying to clarify a doubt. Since, the graph has weights based on\"Fullscreen\", we can assume that when the student is stuck, they try to exit the full screen mode so that they can ask questions from the moderator","metadata":{}},{"cell_type":"code","source":"print(\"The degree of the graph is distributed in the following way:\")\nprint('\\n')\ntextAnalysis.degs()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.216939Z","iopub.execute_input":"2023-06-06T18:44:47.217815Z","iopub.status.idle":"2023-06-06T18:44:47.223906Z","shell.execute_reply.started":"2023-06-06T18:44:47.217754Z","shell.execute_reply":"2023-06-06T18:44:47.222812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above information tells us how important a node is in the graph based on the number of edges coming out of a node. Clearly \"Click\" is the mode important node","metadata":{}},{"cell_type":"code","source":"textAnalysis.bet_centrality()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.225420Z","iopub.execute_input":"2023-06-06T18:44:47.226088Z","iopub.status.idle":"2023-06-06T18:44:47.405160Z","shell.execute_reply.started":"2023-06-06T18:44:47.226048Z","shell.execute_reply":"2023-06-06T18:44:47.404048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"textAnalysis.close_centrality()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.409504Z","iopub.execute_input":"2023-06-06T18:44:47.410250Z","iopub.status.idle":"2023-06-06T18:44:47.567900Z","shell.execute_reply.started":"2023-06-06T18:44:47.410204Z","shell.execute_reply":"2023-06-06T18:44:47.566476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Closeness centrality tells us how fast the flow of information is through a particular node.","metadata":{}},{"cell_type":"code","source":"textAnalysis.shortestPath()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.569075Z","iopub.execute_input":"2023-06-06T18:44:47.569471Z","iopub.status.idle":"2023-06-06T18:44:47.587524Z","shell.execute_reply.started":"2023-06-06T18:44:47.569434Z","shell.execute_reply":"2023-06-06T18:44:47.586219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The average number of steps along the shortest paths for all possible pairs of network nodes in our Graph is 1.75.\n\nThe above information tells us the path from a node to all the other paths and the above information tells us how sequence of events happened.","metadata":{}},{"cell_type":"code","source":"df_train_obj =  trainDataPreparation()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.589210Z","iopub.execute_input":"2023-06-06T18:44:47.590528Z","iopub.status.idle":"2023-06-06T18:44:47.595914Z","shell.execute_reply.started":"2023-06-06T18:44:47.590467Z","shell.execute_reply":"2023-06-06T18:44:47.594741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train_obj.merge()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.597859Z","iopub.execute_input":"2023-06-06T18:44:47.598555Z","iopub.status.idle":"2023-06-06T18:44:47.910056Z","shell.execute_reply.started":"2023-06-06T18:44:47.598514Z","shell.execute_reply":"2023-06-06T18:44:47.908893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['level_group'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:47.913206Z","iopub.execute_input":"2023-06-06T18:44:47.914826Z","iopub.status.idle":"2023-06-06T18:44:48.009483Z","shell.execute_reply.started":"2023-06-06T18:44:47.914773Z","shell.execute_reply":"2023-06-06T18:44:48.008145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictive modeling\n\n**We are moving ahead with an ensemble technique. This would be a custom enseble machine learning technique.**\n\nWe will perform the following tasks:\n1. Clustering: KMeans\n2. Binary classification: tree based model","metadata":{}},{"cell_type":"markdown","source":"# Clustering method","metadata":{}},{"cell_type":"markdown","source":"We are applying clustering algorithm to cluster the data that looks similar. based on the above information, we can say that let's cluster the data into two groups since we have three level groups","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.cluster import KMeans\nimport numpy as np\n\nclass clusters:\n    def __init__(self,df):\n        self.df = df\n        \n    def encode(self):\n        # One-hot encode the categorical variables\n        encoder = OneHotEncoder()\n        encoded_data = encoder.fit_transform(self.df)\n        return encoded_data\n    def clusteringKMeans(self):\n        \n        newEncodedData = self.encode()\n        \n        # Apply k-means clustering on the encoded data\n        kmeans = KMeans(n_clusters=2, random_state=42)\n        clusters = kmeans.fit_predict(newEncodedData)\n\n        # Add cluster labels to the DataFrame\n        self.df['cluster_label'] = clusters\n        \n        return self.df\n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:48.011013Z","iopub.execute_input":"2023-06-06T18:44:48.012208Z","iopub.status.idle":"2023-06-06T18:44:48.422069Z","shell.execute_reply.started":"2023-06-06T18:44:48.012152Z","shell.execute_reply":"2023-06-06T18:44:48.420795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustered_data = clusters(df_train)\nclusteringImplementedData = clustered_data.clusteringKMeans()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:44:48.424000Z","iopub.execute_input":"2023-06-06T18:44:48.424364Z","iopub.status.idle":"2023-06-06T18:45:13.654786Z","shell.execute_reply.started":"2023-06-06T18:44:48.424329Z","shell.execute_reply":"2023-06-06T18:45:13.653728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData['cluster_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.656410Z","iopub.execute_input":"2023-06-06T18:45:13.657159Z","iopub.status.idle":"2023-06-06T18:45:13.673779Z","shell.execute_reply.started":"2023-06-06T18:45:13.657117Z","shell.execute_reply":"2023-06-06T18:45:13.672421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.675508Z","iopub.execute_input":"2023-06-06T18:45:13.676135Z","iopub.status.idle":"2023-06-06T18:45:13.710294Z","shell.execute_reply.started":"2023-06-06T18:45:13.676093Z","shell.execute_reply":"2023-06-06T18:45:13.709148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData_clusterOne = clusteringImplementedData.loc[clusteringImplementedData['cluster_label'] == 0]","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.711875Z","iopub.execute_input":"2023-06-06T18:45:13.712224Z","iopub.status.idle":"2023-06-06T18:45:13.760855Z","shell.execute_reply.started":"2023-06-06T18:45:13.712190Z","shell.execute_reply":"2023-06-06T18:45:13.759284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData_clusterOne.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.762248Z","iopub.execute_input":"2023-06-06T18:45:13.762783Z","iopub.status.idle":"2023-06-06T18:45:13.795955Z","shell.execute_reply.started":"2023-06-06T18:45:13.762746Z","shell.execute_reply":"2023-06-06T18:45:13.794576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total number of level present in first cluster:\")\nprint(clusteringImplementedData_clusterOne['level_group'].value_counts())\nprint('\\n')\nprint(\"Events that dominated in the first cluster looks similar:\")\nprint(clusteringImplementedData_clusterOne['event_name'].value_counts())\nprint('\\n')\nprint('Questions which had similar number of outcome in terms of student performance are:')\nclusteringImplementedData_clusterOne['question'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.797382Z","iopub.execute_input":"2023-06-06T18:45:13.797863Z","iopub.status.idle":"2023-06-06T18:45:13.873470Z","shell.execute_reply.started":"2023-06-06T18:45:13.797809Z","shell.execute_reply":"2023-06-06T18:45:13.872514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData_clusterOne = clusteringImplementedData.loc[clusteringImplementedData['cluster_label'] == 1]\nprint(\"Total number of level present in second cluster:\")\nprint(clusteringImplementedData_clusterOne['level_group'].value_counts())\nprint('\\n')\nprint(\"Events that dominated in the second cluster looks similar:\")\nprint(clusteringImplementedData_clusterOne['event_name'].value_counts())\nprint('\\n')\nprint('Questions which had similar number of outcome in terms of student performance are:')\nclusteringImplementedData_clusterOne['question'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.875018Z","iopub.execute_input":"2023-06-06T18:45:13.875688Z","iopub.status.idle":"2023-06-06T18:45:13.957047Z","shell.execute_reply.started":"2023-06-06T18:45:13.875647Z","shell.execute_reply":"2023-06-06T18:45:13.955936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification","metadata":{}},{"cell_type":"code","source":"# # Rename the duplicate column\n# clusteringImplementedData = clusteringImplementedData.rename(columns={'session_id': 'session_id_1'})\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.958594Z","iopub.execute_input":"2023-06-06T18:45:13.959287Z","iopub.status.idle":"2023-06-06T18:45:13.964498Z","shell.execute_reply.started":"2023-06-06T18:45:13.959244Z","shell.execute_reply":"2023-06-06T18:45:13.962712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.966315Z","iopub.execute_input":"2023-06-06T18:45:13.966705Z","iopub.status.idle":"2023-06-06T18:45:13.985227Z","shell.execute_reply.started":"2023-06-06T18:45:13.966665Z","shell.execute_reply":"2023-06-06T18:45:13.984115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData.columns = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level',\n       'page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n       'hover_duration', 'text', 'fqid', 'room_fqid', 'text_fqid',\n       'fullscreen', 'hq', 'music', 'level_group', 'event_id',\n       'event_name_activity', 'Source_questionMark', 'session_id_1_to_remove', 'correct',\n       'Session_ID_labels', 'question', 'cluster_label']","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:13.995876Z","iopub.execute_input":"2023-06-06T18:45:13.998256Z","iopub.status.idle":"2023-06-06T18:45:14.005938Z","shell.execute_reply.started":"2023-06-06T18:45:13.998195Z","shell.execute_reply":"2023-06-06T18:45:14.004243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData = clusteringImplementedData.drop('session_id_1_to_remove', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:14.008073Z","iopub.execute_input":"2023-06-06T18:45:14.008484Z","iopub.status.idle":"2023-06-06T18:45:14.051336Z","shell.execute_reply.started":"2023-06-06T18:45:14.008445Z","shell.execute_reply":"2023-06-06T18:45:14.050148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:14.052934Z","iopub.execute_input":"2023-06-06T18:45:14.053409Z","iopub.status.idle":"2023-06-06T18:45:14.063351Z","shell.execute_reply.started":"2023-06-06T18:45:14.053370Z","shell.execute_reply":"2023-06-06T18:45:14.062111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusteringImplementedData.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:14.064992Z","iopub.execute_input":"2023-06-06T18:45:14.065492Z","iopub.status.idle":"2023-06-06T18:45:14.075054Z","shell.execute_reply.started":"2023-06-06T18:45:14.065454Z","shell.execute_reply":"2023-06-06T18:45:14.073977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:14.076649Z","iopub.execute_input":"2023-06-06T18:45:14.077121Z","iopub.status.idle":"2023-06-06T18:45:24.612648Z","shell.execute_reply.started":"2023-06-06T18:45:14.077085Z","shell.execute_reply":"2023-06-06T18:45:24.610985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels['session'] = df_train_labels.session_id.apply(lambda x: int(x.split('_')[0]) )\ndf_train_labels['q'] = df_train_labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:24.614297Z","iopub.execute_input":"2023-06-06T18:45:24.615121Z","iopub.status.idle":"2023-06-06T18:45:25.817167Z","shell.execute_reply.started":"2023-06-06T18:45:24.615072Z","shell.execute_reply":"2023-06-06T18:45:25.816096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:25.819096Z","iopub.execute_input":"2023-06-06T18:45:25.819858Z","iopub.status.idle":"2023-06-06T18:45:25.840559Z","shell.execute_reply.started":"2023-06-06T18:45:25.819795Z","shell.execute_reply":"2023-06-06T18:45:25.839625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:45:25.842206Z","iopub.execute_input":"2023-06-06T18:45:25.842907Z","iopub.status.idle":"2023-06-06T18:45:25.848210Z","shell.execute_reply.started":"2023-06-06T18:45:25.842834Z","shell.execute_reply":"2023-06-06T18:45:25.847052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:21.400490Z","iopub.execute_input":"2023-06-06T18:46:21.401018Z","iopub.status.idle":"2023-06-06T18:46:21.412140Z","shell.execute_reply.started":"2023-06-06T18:46:21.400973Z","shell.execute_reply":"2023-06-06T18:46:21.411049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = feature_engineer(clusteringImplementedData)\nprint(\"Full prepared dataset shape is {}\".format(clusteringImplementedData.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:22.486248Z","iopub.execute_input":"2023-06-06T18:46:22.486686Z","iopub.status.idle":"2023-06-06T18:46:23.016111Z","shell.execute_reply.started":"2023-06-06T18:46:22.486639Z","shell.execute_reply":"2023-06-06T18:46:23.014898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:25.387601Z","iopub.execute_input":"2023-06-06T18:46:25.388074Z","iopub.status.idle":"2023-06-06T18:46:25.421537Z","shell.execute_reply.started":"2023-06-06T18:46:25.388026Z","shell.execute_reply":"2023-06-06T18:46:25.420546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:29.015754Z","iopub.execute_input":"2023-06-06T18:46:29.016197Z","iopub.status.idle":"2023-06-06T18:46:29.029617Z","shell.execute_reply.started":"2023-06-06T18:46:29.016155Z","shell.execute_reply":"2023-06-06T18:46:29.028517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = tfdf.keras.GradientBoostedTreesModel(hyperparameter_template=\"benchmark_rank1\")","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:33.766191Z","iopub.execute_input":"2023-06-06T18:46:33.766647Z","iopub.status.idle":"2023-06-06T18:46:34.198682Z","shell.execute_reply.started":"2023-06-06T18:46:33.766606Z","shell.execute_reply":"2023-06-06T18:46:34.197710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nVALID_USER_LIST = valid_x.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:37.046987Z","iopub.execute_input":"2023-06-06T18:46:37.048148Z","iopub.status.idle":"2023-06-06T18:46:37.055646Z","shell.execute_reply.started":"2023-06-06T18:46:37.048093Z","shell.execute_reply":"2023-06-06T18:46:37.054610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n        \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = train_x.loc[train_x.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp]\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = df_train_labels.loc[df_train_labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = df_train_labels.loc[df_train_labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # There's one more step required before we can train the model. \n    # We need to convert the datatset from Pandas format (pd.DataFrame)\n    # into TensorFlow Datasets format (tf.data.Dataset).\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n\n    # We will now create the Gradient Boosted Trees Model with default settings. \n    # By default the model is set to train for a classification task.\n    gbtm = tfdf.keras.GradientBoostedTreesModel(verbose=0)\n    gbtm.compile(metrics=[\"accuracy\"])\n\n    # Train the model.\n    gbtm.fit(x=train_ds)\n\n    # Store the model\n    models[f'{grp}_{q_no}'] = gbtm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    inspector = gbtm.make_inspector()\n    inspector.evaluation()\n    evaluation = gbtm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]         \n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = gbtm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()   ","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:46:52.682162Z","iopub.execute_input":"2023-06-06T18:46:52.682623Z","iopub.status.idle":"2023-06-06T18:47:29.242950Z","shell.execute_reply.started":"2023-06-06T18:46:52.682578Z","shell.execute_reply":"2023-06-06T18:47:29.241997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:48:35.917420Z","iopub.execute_input":"2023-06-06T18:48:35.918694Z","iopub.status.idle":"2023-06-06T18:48:35.925496Z","shell.execute_reply.started":"2023-06-06T18:48:35.918637Z","shell.execute_reply":"2023-06-06T18:48:35.924509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.model_plotter.plot_model_in_colab(models['0-4_1'], tree_idx=0, max_depth=3)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:48:39.849933Z","iopub.execute_input":"2023-06-06T18:48:39.850468Z","iopub.status.idle":"2023-06-06T18:48:39.866789Z","shell.execute_reply.started":"2023-06-06T18:48:39.850425Z","shell.execute_reply":"2023-06-06T18:48:39.865246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe of required size:\n# (no: of users in validation set x no: of questions) initialized to zero values\n# to store true values of the label `correct`. \ntrue_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    # Get the true labels.\n    tmp = df_train_labels.loc[df_train_labels.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0\n\n# Loop through threshold values from 0.4 to 0.8 and select the threshold with \n# the highest `F1 score`.\nfor threshold in np.arange(0.4,0.8,0.01):\n    metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    if f1_score > max_score:\n        max_score = f1_score\n        best_threshold = threshold\n        \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:48:43.818754Z","iopub.execute_input":"2023-06-06T18:48:43.819224Z","iopub.status.idle":"2023-06-06T18:48:44.935241Z","shell.execute_reply.started":"2023-06-06T18:48:43.819179Z","shell.execute_reply":"2023-06-06T18:48:44.934046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\n\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:48:47.555897Z","iopub.execute_input":"2023-06-06T18:48:47.556350Z","iopub.status.idle":"2023-06-06T18:48:47.574337Z","shell.execute_reply.started":"2023-06-06T18:48:47.556307Z","shell.execute_reply":"2023-06-06T18:48:47.572792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        test_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, test_df.columns != 'level_group'])\n        predictions = gbtm.predict(test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:48:48.883926Z","iopub.execute_input":"2023-06-06T18:48:48.885284Z","iopub.status.idle":"2023-06-06T18:48:54.817371Z","shell.execute_reply.started":"2023-06-06T18:48:48.885232Z","shell.execute_reply":"2023-06-06T18:48:54.816216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-06T18:49:14.075746Z","iopub.execute_input":"2023-06-06T18:49:14.077013Z","iopub.status.idle":"2023-06-06T18:49:15.273222Z","shell.execute_reply.started":"2023-06-06T18:49:14.076955Z","shell.execute_reply":"2023-06-06T18:49:15.271919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}