{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"One of the key aspects of TensorFlow Decision Forests that makes it even more suitable for this competition, particularly given the runtime limitations, is that it has been extensively tested for training and inference on CPUs, making it possible to train it on lower-end machines.","metadata":{}},{"cell_type":"markdown","source":"# Import the Required Libraries","metadata":{"id":"zAXHC6-Tn2O5"}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"id":"IanlX-Eqn2O5","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    train_csv_path = \"/kaggle/input/predict-student-performance-from-game-play/train.csv\"\n    train_labels_path = \"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\"\n    \n    test_ratio = 0.20\n    verbose = True # Tensorflow verbosity\n    model = 'gbtm'\n    questions = 3 # 3 - all, 2 - final group, 1 - second group, 0 - first group","metadata":{"id":"gLpK2yAen2O7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"id":"_XItl24kn2O7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df = pd.read_csv(Config.train_labels_path)\n# Separate the session id into the session and the question number\nlabels_df['session'] = labels_df['session_id'].apply(lambda x: int(x.split(\"_\")[0]))\nlabels_df['q'] = labels_df['session_id'].apply(lambda x: int(x.split(\"_\")[-1][1:]))","metadata":{"id":"KD4uayl2n2O9","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DatasetUtils:\n    def __init__(self, df):\n        self.df = df\n    \n    def feature_engineer(self):\n        # Reference: https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n        dfs = []\n        for c in CATEGORICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('nunique')\n            tmp.name = tmp.name + '_nunique'\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('mean')\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('std')\n            tmp.name = tmp.name + '_std'\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg(np.var)\n            tmp.name = tmp.name + '_var'\n            dfs.append(tmp)\n        self.df = pd.concat(dfs,axis=1)\n        self.df = self.df.fillna(-1)\n        self.df = self.df.reset_index()\n        self.df = self.df.set_index('session_id')\n        print(f\"Full prepared dataset shape is {self.df.shape}\")\n        return self.df\n    \n    def split_dataset(self, test_ratio=Config.test_ratio):\n        USER_LIST = self.df.index.unique()\n        split = int(len(USER_LIST) * (1 - test_ratio))\n        return self.df.loc[USER_LIST[:split]], self.df.loc[USER_LIST[split:]]class DatasetUtils:\n    def __init__(self, df):\n        self.df = df\n    \n    def feature_engineer(self):\n        # Reference: https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n        dfs = []\n        for c in CATEGORICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('nunique')\n            tmp.name = tmp.name + '_nunique'\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('mean')\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('std')\n            tmp.name = tmp.name + '_std'\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg(np.var)\n            tmp.name = tmp.name + '_var'\n            dfs.append(tmp)\n        self.df = pd.concat(dfs,axis=1)\n        self.df = self.df.fillna(-1)\n        self.df = self.df.reset_index()\n        self.df = self.df.set_index('session_id')\n        print(f\"Full prepared dataset shape is {self.df.shape}\")\n        return self.df\n    \n    def split_dataset(self, test_ratio=Config.test_ratio):\n        USER_LIST = self.df.index.unique()\n        split = int(len(USER_LIST) * (1 - test_ratio))\n        return self.df.loc[USER_LIST[:split]], self.df.loc[USER_LIST[split:]]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = DatasetUtils(dataset_df)\nds.feature_engineer()\ntrain_x, valid_x = ds.split_dataset()\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ModelHelper:\n    def __init__(self, train, val, labels, tuner):\n        self.train = train\n        self.valid = val\n        self.labels = labels\n        \n        self.verbose = Config.verbose\n        \n        self.VALID_USER_LIST = val.index.unique()\n        self.prediction_df = pd.DataFrame(data=np.zeros((len(self.VALID_USER_LIST), 18)), index=self.VALID_USER_LIST)\n        \n        self.question_groups = ['0-4', '5-12', '13-22']\n        self.model = None\n        self.model_dict = {}\n        self.eval_dict = {}\n        \n    def build_model(self, model=Config.model):\n        \"\"\"\n        Builds the tensorflow model specified, sets the instance model to the selected model\n        Returns the model, (return capture not required)\n        \"\"\"\n        # build and return the specified model\n        match model:\n            case 'gbtm':\n                self.model = tfdf.keras.GradientBoostedTreesModel(num_trees=50, verbose=self.verbose, tuner=tuner)\n                self.model.compile(metrics=[\"accuracy\"])\n                return model\n                \n    def train_model(self, questions=3):\n        \"\"\"\n        Iterate through the specified question range\n        \"\"\"\n        # default is full range\n        # if needed to target certain questions \n        match questions:\n            case 3: # full range\n                q_range = range(1, 19)\n            case 2: # final group\n                q_range = range(14, 19)\n            case 1: # second group\n                q_range = range(5, 14)\n            case 0: # first group\n                q_range = range(1, 4)\n                \n        for q_no in q_range:\n            if q_no <= 3: grp = self.question_groups[0]\n            elif q_no <= 13: grp = self.question_groups[1]\n            elif q_no <= 22: grp = self.question_groups[2]\n            \n            if self.verbose:\n                print(\"### q_no: {} -- grp: {}\".format(q_no, grp))\n            \n            # Find the selected group in the dataset\n            train_df = self.train.loc[self.train.level_group == grp]\n            train_users = train_df.index.values\n            valid_df = self.valid.loc[self.valid.level_group == grp]\n            valid_users = valid_df.index.values\n            \n            # Find the questions based on the group number in the labels\n            train_labels = self.labels.loc[self.labels.q == q_no].set_index('session').loc[train_users]\n            valid_labels = self.labels.loc[self.labels.q == q_no].set_index('session').loc[valid_users]\n            \n            # add the target to the train dataset\n            train_df.loc[:, 'target'] = train_labels.loc[:, 'correct']\n            valid_df.loc[:, 'target'] = valid_labels.loc[:, 'correct']\n            \n            # Convert the pandas dataframe to a tensorflow tensor-type dataset\n            train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"target\")\n            valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"target\")\n            \n            # Set the model and fit to the training data\n            model = self.model\n            model.fit(x=train_ds)\n            \n            # store the model\n            self.model_dict[f'{grp}_{q_no}'] = model\n            \n            # Evaluate the trained model on the validation dataset and store the \n            # evaluation accuracy in the `evaluation_dict`.\n            inspector = model.make_inspector()\n            inspector.evaluation()\n            evaluation = model.evaluate(x=valid_ds,return_dict=True)\n            self.eval_dict[q_no] = evaluation[\"accuracy\"]\n            \n            # Use the trained model to make predictions on the validation dataset and \n            # store the predicted values in the `prediction_df` dataframe.\n            predict = model.predict(x=valid_ds)\n            self.prediction_df.loc[valid_users, q_no-1] = predict.flatten()\n    \n    def predict(self):\n        # Create a dataframe of required size:\n        # (no: of users in validation set x no: of questions) initialized to zero values\n        # to store true values of the label `correct`. \n        true_df = pd.DataFrame(data=np.zeros((len(self.VALID_USER_LIST),18)), index=self.VALID_USER_LIST)\n        for i in range(18):\n            # Get the true labels.\n            tmp = self.labels.loc[self.labels.q == i+1].set_index('session').loc[self.VALID_USER_LIST]\n            true_df[i] = tmp.correct.values\n\n        max_score = 0; best_threshold = 0\n\n        # Loop through threshold values from 0.4 to 0.8 and select the threshold with \n        # the highest `F1 score`.\n        for threshold in np.arange(0.4,0.8,0.01):\n            metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n            y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n            y_pred = tf.one_hot((self.prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n            metric.update_state(y_true, y_pred)\n            f1_score = metric.result().numpy()\n            if f1_score > max_score:\n                max_score = f1_score\n                best_threshold = threshold\n\n        print(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tuner = tfdf.tuner.RandomSearch(num_trials=20)\ntuner.choice(\"num_candidate_attributes_ratio\", [1.0, 0.8, 0.6])\ntuner.choice(\"use_hessian_gain\", [True, False])\n\nlocal_search_space = tuner.choice(\"growing_strategy\", [\"LOCAL\"])\nlocal_search_space.choice(\"max_depth\", [4, 5, 6, 7])\n\nglobal_search_space = tuner.choice(\n    \"growing_strategy\", [\"BEST_FIRST_GLOBAL\"], merge=True)\nglobal_search_space.choice(\"max_num_nodes\", [16, 32, 64, 128])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiate the model helper, build and train the model\nmh = ModelHelper(train_x, valid_x, labels_df, tuner)\nmh.build_model()\nmh.train_model()\n\n# access the objects models saved and evaluation dict\nmodels = mh.model_dict\nevaluation_dict = mh.eval_dict\n\nfor name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mh.predict()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\n\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        test_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, test_df.columns != 'level_group'])\n        predictions = gbtm.predict(test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{},"execution_count":null,"outputs":[]}]}