{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries and set Configuration","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-08T01:44:38.617476Z","iopub.execute_input":"2023-05-08T01:44:38.617843Z","iopub.status.idle":"2023-05-08T01:44:47.477238Z","shell.execute_reply.started":"2023-05-08T01:44:38.617814Z","shell.execute_reply":"2023-05-08T01:44:47.476125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    train_csv_path = \"/kaggle/input/predict-student-performance-from-game-play/train.csv\"\n    train_labels_path = \"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\"\n    \n    test_ratio = 0.20\n    verbose = False # Tensorflow verbosity\n    model = 'gbtm'\n    questions = 3 # 3 - all, 2 - final group, 1 - second group, 0 - first group","metadata":{"execution":{"iopub.status.busy":"2023-05-08T01:44:47.478713Z","iopub.execute_input":"2023-05-08T01:44:47.479662Z","iopub.status.idle":"2023-05-08T01:44:47.484512Z","shell.execute_reply.started":"2023-05-08T01:44:47.479630Z","shell.execute_reply":"2023-05-08T01:44:47.483525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import the dataset and set the datatypes","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\n# dataset_df = pd.read_csv(Config.train_csv_path, dtype=dtypes)\ndataset_df = pd.read_csv(\"/kaggle/input/condensed-df-student-game-learning/condensed_df.csv\")\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:34:23.378643Z","iopub.execute_input":"2023-05-08T02:34:23.379020Z","iopub.status.idle":"2023-05-08T02:34:23.694666Z","shell.execute_reply.started":"2023-05-08T02:34:23.378995Z","shell.execute_reply":"2023-05-08T02:34:23.693778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import random\n# random_choices = random.choices(range(len(dataset_df)), k=100000)\n# random_choices.sort()\n# random_df = dataset_df.iloc[random_choices]\n\n# random_df.to_csv('/kaggle/working/condensed_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:11:43.189563Z","iopub.execute_input":"2023-05-08T02:11:43.190016Z","iopub.status.idle":"2023-05-08T02:11:43.193144Z","shell.execute_reply.started":"2023-05-08T02:11:43.189990Z","shell.execute_reply":"2023-05-08T02:11:43.192472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df = pd.read_csv(Config.train_labels_path)\n# Separate the session id into the session and the question number\nlabels_df['session'] = labels_df['session_id'].apply(lambda x: int(x.split(\"_\")[0]))\nlabels_df['q'] = labels_df['session_id'].apply(lambda x: int(x.split(\"_\")[-1][1:]))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:11:43.218263Z","iopub.execute_input":"2023-05-08T02:11:43.218818Z","iopub.status.idle":"2023-05-08T02:11:44.288701Z","shell.execute_reply.started":"2023-05-08T02:11:43.218788Z","shell.execute_reply":"2023-05-08T02:11:44.287831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df[\"composite_coor\"] = (\n    dataset_df[\"room_coor_x\"] + \n    dataset_df[\"room_coor_y\"] + \n    dataset_df[\"screen_coor_x\"] + \n    dataset_df[\"screen_coor_y\"]\n) / 4\n\ndataset_df['x_coor_diff'] = abs(dataset_df['room_coor_x'] - dataset_df['screen_coor_x'])\ndataset_df['y_coor_diff'] = abs(dataset_df['room_coor_y'] - dataset_df['screen_coor_y'])","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:34:25.449575Z","iopub.execute_input":"2023-05-08T02:34:25.449916Z","iopub.status.idle":"2023-05-08T02:34:25.464488Z","shell.execute_reply.started":"2023-05-08T02:34:25.449890Z","shell.execute_reply":"2023-05-08T02:34:25.463473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_sum = dataset_df.groupby('level')['composite_coor'].std()\nx_diff_mean = dataset_df.groupby('level')['x_coor_diff'].mean()\ny_diff_mean = dataset_df.groupby('level')['y_coor_diff'].mean()\nplt.figure()\n\nplt.bar(range(len(std_sum)), std_level)\nplt.title(\"Standard Deviation of composite coor by question\")","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:40:51.586576Z","iopub.execute_input":"2023-05-08T02:40:51.586935Z","iopub.status.idle":"2023-05-08T02:40:51.815995Z","shell.execute_reply.started":"2023-05-08T02:40:51.586908Z","shell.execute_reply":"2023-05-08T02:40:51.815171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracies = [0.7286, 0.9756, 0.9351, 0.7952, 0.6306, 0.7889, 0.7481, 0.6321, 0.7677, 0.6011, 0.6550, 0.8697, 0.7225, 0.7316, 0.6119, 0.7479, 0.7034, 0.9512, None, None, None, None, None]\nstd_level_df = pd.DataFrame(std_sum, std_level_dct.keys())\nstd_level_df['accuracy'] = accuracies\nstd_level_df['x_coor'] = x_diff_mean\nstd_level_df['y_coor'] = y_diff_mean\nstd_level_df['level'] = list(range(0, 23))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:41:10.068669Z","iopub.execute_input":"2023-05-08T02:41:10.069012Z","iopub.status.idle":"2023-05-08T02:41:10.078500Z","shell.execute_reply.started":"2023-05-08T02:41:10.068987Z","shell.execute_reply":"2023-05-08T02:41:10.077275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_level_df.iloc[:18].corr()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:41:29.937620Z","iopub.execute_input":"2023-05-08T02:41:29.937969Z","iopub.status.idle":"2023-05-08T02:41:29.956246Z","shell.execute_reply.started":"2023-05-08T02:41:29.937942Z","shell.execute_reply":"2023-05-08T02:41:29.955114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_level_df.corr()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:29:40.199363Z","iopub.execute_input":"2023-05-08T02:29:40.199742Z","iopub.status.idle":"2023-05-08T02:29:40.215074Z","shell.execute_reply.started":"2023-05-08T02:29:40.199717Z","shell.execute_reply":"2023-05-08T02:29:40.214226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare the Dataset and Engineer the features and Train/Test Split","metadata":{}},{"cell_type":"code","source":"class DatasetUtils:\n    def __init__(self, df):\n        self.df = df\n    \n    def feature_engineer(self):\n        # Reference: https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n        dfs = []\n        for c in CATEGORICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('nunique')\n            tmp.name = tmp.name + '_nunique'\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('mean')\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg('std')\n            tmp.name = tmp.name + '_std'\n            dfs.append(tmp)\n        for c in NUMERICAL:\n            tmp = self.df.groupby(['session_id','level_group'])[c].agg(np.var)\n            tmp.name = tmp.name + '_var'\n            dfs.append(tmp)\n        self.df = pd.concat(dfs,axis=1)\n        self.df = self.df.fillna(-1)\n        self.df = self.df.reset_index()\n        self.df = self.df.set_index('session_id')\n        print(f\"Full prepared dataset shape is {self.df.shape}\")\n        return self.df\n    \n    def split_dataset(self, test_ratio=Config.test_ratio):\n        USER_LIST = self.df.index.unique()\n        split = int(len(USER_LIST) * (1 - test_ratio))\n        return self.df.loc[USER_LIST[:split]], self.df.loc[USER_LIST[split:]]\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:26:00.386613Z","iopub.execute_input":"2023-05-08T02:26:00.386989Z","iopub.status.idle":"2023-05-08T02:26:00.404364Z","shell.execute_reply.started":"2023-05-08T02:26:00.386962Z","shell.execute_reply":"2023-05-08T02:26:00.403456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid', 'level']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration', 'accuracy']","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:26:00.500928Z","iopub.execute_input":"2023-05-08T02:26:00.501507Z","iopub.status.idle":"2023-05-08T02:26:00.505749Z","shell.execute_reply.started":"2023-05-08T02:26:00.501475Z","shell.execute_reply":"2023-05-08T02:26:00.504953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = DatasetUtils(dataset_df)\nds.feature_engineer()\ntrain_x, valid_x = ds.split_dataset()\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:26:01.086273Z","iopub.execute_input":"2023-05-08T02:26:01.086869Z","iopub.status.idle":"2023-05-08T02:26:02.485314Z","shell.execute_reply.started":"2023-05-08T02:26:01.086837Z","shell.execute_reply":"2023-05-08T02:26:02.482094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds.df.corr()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T02:28:00.682832Z","iopub.execute_input":"2023-05-08T02:28:00.683190Z","iopub.status.idle":"2023-05-08T02:28:00.881209Z","shell.execute_reply.started":"2023-05-08T02:28:00.683164Z","shell.execute_reply":"2023-05-08T02:28:00.880478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train the model","metadata":{}},{"cell_type":"code","source":"class ModelHelper:\n    def __init__(self, train, val, labels, tuner):\n        self.train = train\n        self.valid = val\n        self.labels = labels\n        \n        self.verbose = Config.verbose\n        \n        self.VALID_USER_LIST = val.index.unique()\n        self.prediction_df = pd.DataFrame(data=np.zeros((len(self.VALID_USER_LIST), 18)), index=self.VALID_USER_LIST)\n        \n        self.question_groups = ['0-4', '5-12', '13-22']\n        self.model = None\n        self.model_dict = {}\n        self.eval_dict = {}\n        \n        self.question_ranges = {0: range(1, 4), 1: range(5, 14), 2: range(14, 19), 3: range(1, 19)}\n        \n    def build_model(self, model=Config.model):\n        \"\"\"\n        Builds the tensorflow model specified, sets the instance model to the selected model\n        Returns the model, (return capture not required)\n        \"\"\"\n        # build and return the specified model\n        if model == \"gbtm\":\n            self.model = tfdf.keras.GradientBoostedTreesModel(num_trees=50, verbose=self.verbose, tuner=tuner)\n            self.model.compile(metrics=[\"accuracy\"])\n        return self.model\n                \n    def train_model(self, questions=3):\n        \"\"\"\n        Iterate through the specified question range\n        \"\"\"\n        # default is full range\n        # if needed to target certain questions \n        q_range = self.question_ranges[questions]\n                \n        for q_no in q_range:\n            if q_no <= 3: grp = self.question_groups[0]\n            elif q_no <= 13: grp = self.question_groups[1]\n            elif q_no <= 22: grp = self.question_groups[2]\n            \n            if self.verbose:\n                print(\"### q_no: {} -- grp: {}\".format(q_no, grp))\n            \n            # Find the selected group in the dataset\n            train_df = self.train.loc[self.train.level_group == grp]\n            train_users = train_df.index.values\n            valid_df = self.valid.loc[self.valid.level_group == grp]\n            valid_users = valid_df.index.values\n            \n            # Find the questions based on the group number in the labels\n            train_labels = self.labels.loc[self.labels.q == q_no].set_index('session').loc[train_users]\n            valid_labels = self.labels.loc[self.labels.q == q_no].set_index('session').loc[valid_users]\n            \n            # add the target to the train dataset\n            train_df.loc[:, 'target'] = train_labels.loc[:, 'correct']\n            valid_df.loc[:, 'target'] = valid_labels.loc[:, 'correct']\n            \n            # Convert the pandas dataframe to a tensorflow tensor-type dataset\n            train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"target\")\n            valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"target\")\n            \n            # Set the model and fit to the training data\n            model = self.model\n            model.fit(x=train_ds)\n            \n            # store the model\n            self.model_dict[f'{grp}_{q_no}'] = model\n            \n            # Evaluate the trained model on the validation dataset and store the \n            # evaluation accuracy in the `evaluation_dict`.\n            inspector = model.make_inspector()\n            inspector.evaluation()\n            evaluation = model.evaluate(x=valid_ds,return_dict=True)\n            self.eval_dict[q_no] = evaluation[\"accuracy\"]\n            \n            # Use the trained model to make predictions on the validation dataset and \n            # store the predicted values in the `prediction_df` dataframe.\n            predict = model.predict(x=valid_ds)\n            self.prediction_df.loc[valid_users, q_no-1] = predict.flatten()\n    \n    def predict(self):\n        # Create a dataframe of required size:\n        # (no: of users in validation set x no: of questions) initialized to zero values\n        # to store true values of the label `correct`. \n        true_df = pd.DataFrame(data=np.zeros((len(self.VALID_USER_LIST),18)), index=self.VALID_USER_LIST)\n        for i in range(18):\n            # Get the true labels.\n            tmp = self.labels.loc[self.labels.q == i+1].set_index('session').loc[self.VALID_USER_LIST]\n            true_df[i] = tmp.correct.values\n\n        max_score = 0; best_threshold = 0\n\n        # Loop through threshold values from 0.4 to 0.8 and select the threshold with \n        # the highest `F1 score`.\n        for threshold in np.arange(0.4,0.8,0.01):\n            metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n            y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n            y_pred = tf.one_hot((self.prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n            metric.update_state(y_true, y_pred)\n            f1_score = metric.result().numpy()\n            if f1_score > max_score:\n                max_score = f1_score\n                best_threshold = threshold\n\n        print(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T20:55:52.638525Z","iopub.execute_input":"2023-05-06T20:55:52.639020Z","iopub.status.idle":"2023-05-06T20:55:52.661497Z","shell.execute_reply.started":"2023-05-06T20:55:52.638989Z","shell.execute_reply":"2023-05-06T20:55:52.660607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tuner = tfdf.tuner.RandomSearch(num_trials=20)\ntuner.choice(\"num_candidate_attributes_ratio\", [1.0, 0.8, 0.6])\ntuner.choice(\"use_hessian_gain\", [True, False])\n\nlocal_search_space = tuner.choice(\"growing_strategy\", [\"LOCAL\"])\nlocal_search_space.choice(\"max_depth\", [4, 5, 6, 7])\n\nglobal_search_space = tuner.choice(\n    \"growing_strategy\", [\"BEST_FIRST_GLOBAL\"], merge=True)\nglobal_search_space.choice(\"max_num_nodes\", [16, 32, 64, 128])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T20:55:55.153240Z","iopub.execute_input":"2023-05-06T20:55:55.153620Z","iopub.status.idle":"2023-05-06T20:55:55.168441Z","shell.execute_reply.started":"2023-05-06T20:55:55.153592Z","shell.execute_reply":"2023-05-06T20:55:55.167504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiate the model helper, build and train the model\nmh = ModelHelper(train_x, valid_x, labels_df, tuner)\nmh.build_model()\nmh.train_model()\n\n# access the objects models saved and evaluation dict\nmodels = mh.model_dict\nevaluation_dict = mh.eval_dict\n\nfor name, value in evaluation_dict.items():\n    print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T20:56:08.886856Z","iopub.execute_input":"2023-05-06T20:56:08.887262Z","iopub.status.idle":"2023-05-06T21:17:48.426446Z","shell.execute_reply.started":"2023-05-06T20:56:08.887234Z","shell.execute_reply":"2023-05-06T21:17:48.425481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Apply predictions to the test_df","metadata":{}},{"cell_type":"code","source":"mh.predict()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T21:17:48.428095Z","iopub.execute_input":"2023-05-06T21:17:48.428677Z","iopub.status.idle":"2023-05-06T21:17:49.619782Z","shell.execute_reply.started":"2023-05-06T21:17:48.428645Z","shell.execute_reply":"2023-05-06T21:17:49.618819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        test_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, test_df.columns != 'level_group'])\n        predictions = gbtm.predict(test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T20:06:01.472981Z","iopub.execute_input":"2023-05-06T20:06:01.473457Z","iopub.status.idle":"2023-05-06T20:06:01.737726Z","shell.execute_reply.started":"2023-05-06T20:06:01.473413Z","shell.execute_reply":"2023-05-06T20:06:01.736377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-05-04T16:56:28.473779Z","iopub.status.idle":"2023-05-04T16:56:28.474147Z","shell.execute_reply.started":"2023-05-04T16:56:28.473969Z","shell.execute_reply":"2023-05-04T16:56:28.473985Z"},"trusted":true},"execution_count":null,"outputs":[]}]}