{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{"id":"zAXHC6-Tn2O5"}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"id":"IanlX-Eqn2O5","execution":{"iopub.status.busy":"2023-08-03T12:04:02.054443Z","iopub.execute_input":"2023-08-03T12:04:02.055314Z","iopub.status.idle":"2023-08-03T12:04:11.581469Z","shell.execute_reply.started":"2023-08-03T12:04:02.055257Z","shell.execute_reply":"2023-08-03T12:04:11.579941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Dataset","metadata":{"id":"22DpLVFLn2O7"}},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"id":"_XItl24kn2O7","execution":{"iopub.status.busy":"2023-08-03T12:04:11.583184Z","iopub.execute_input":"2023-08-03T12:04:11.583889Z","iopub.status.idle":"2023-08-03T12:06:48.113006Z","shell.execute_reply.started":"2023-08-03T12:04:11.583842Z","shell.execute_reply":"2023-08-03T12:06:48.111575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first 5 examples\ndataset_df.head(5)","metadata":{"id":"-RTRVRiWn2O8","execution":{"iopub.status.busy":"2023-08-03T12:06:48.114987Z","iopub.execute_input":"2023-08-03T12:06:48.115448Z","iopub.status.idle":"2023-08-03T12:06:48.156623Z","shell.execute_reply.started":"2023-08-03T12:06:48.115401Z","shell.execute_reply":"2023-08-03T12:06:48.154859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the labels and split session_id","metadata":{"id":"ReNY-i3bn2O8"}},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"id":"KD4uayl2n2O9","execution":{"iopub.status.busy":"2023-08-03T12:06:48.160495Z","iopub.execute_input":"2023-08-03T12:06:48.161093Z","iopub.status.idle":"2023-08-03T12:06:49.766188Z","shell.execute_reply.started":"2023-08-03T12:06:48.161054Z","shell.execute_reply":"2023-08-03T12:06:49.764968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first 5 examples\nlabels.head(5)","metadata":{"id":"0eD-KZMvn2O-","execution":{"iopub.status.busy":"2023-08-03T12:06:49.767674Z","iopub.execute_input":"2023-08-03T12:06:49.770774Z","iopub.status.idle":"2023-08-03T12:06:49.783241Z","shell.execute_reply.started":"2023-08-03T12:06:49.770715Z","shell.execute_reply":"2023-08-03T12:06:49.782158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing the dataset\n\nWe will ignore fullscreen, hq and music as they are unrelated columns\n\n## Feature engineering","metadata":{"id":"y5fK05dsn2O_"}},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']\nevent_types = list(np.unique(dataset_df[\"event_name\"]))\n# print(event_types)\n\ndef feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    # Make columns for event types, getting their count for a session_id and level_group\n    for event in event_types:\n        tmp = dataset_df.groupby(['session_id', 'level_group'])['event_name'].apply(lambda x: (x == event).sum())\n        tmp.name = event + '_sum'\n        dfs.append(tmp)\n    # Make columns for each event type elapsed time. getting their mean for a session_id and level_group\n    tmp = dataset_df.groupby(['session_id', 'level_group']).apply(lambda x: x[[\"level_group\", \"elapsed_time\"]][\"elapsed_time\"].max())\n    tmp.name = \"elapsed_time_per_level\"\n    dfs.append(tmp)\n\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df\n\ndef outlier_capping(x, upper, lower):\n    if x > upper:\n        return upper\n    if x < lower:\n        return lower\n    else:\n        return x\n\ndef pre_processing(train):\n    # fill empty numerical values with mean value, and add a dummy column indicating it was NaN\n    for column in ['elapsed_time','level','page','room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']:\n        train[column].fillna(train[column].mean(), inplace=True)\n\n    train = train.sort_values(by=['session_id', 'elapsed_time'])\n    train['Time Spent Per Action'] = train.groupby('session_id')['elapsed_time'].diff().fillna(0)    \n    outliers = ['Time Spent Per Action', 'hover_duration']\n    for outlier in outliers:\n        q1, q3 = np.percentile(train[outlier], [25, 75])\n        iqr = q3 - q1\n        upper_boundary = q3 + iqr * 1.5\n        lower_boundary = q1 - iqr * 1.5\n        train[outlier] = train[outlier].apply(lambda x: outlier_capping(x, upper_boundary, lower_boundary))\n    \n    return train","metadata":{"id":"cCZWGiL_n2PA","execution":{"iopub.status.busy":"2023-08-03T12:06:49.785566Z","iopub.execute_input":"2023-08-03T12:06:49.786399Z","iopub.status.idle":"2023-08-03T12:07:33.005791Z","shell.execute_reply.started":"2023-08-03T12:06:49.786321Z","shell.execute_reply":"2023-08-03T12:07:33.004550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = feature_engineer(dataset_df)\ndataset_df = pre_processing(dataset_df)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df.shape))","metadata":{"id":"JKcoPoemn2PA","execution":{"iopub.status.busy":"2023-08-03T12:07:33.008002Z","iopub.execute_input":"2023-08-03T12:07:33.008854Z","iopub.status.idle":"2023-08-03T12:13:58.173792Z","shell.execute_reply.started":"2023-08-03T12:07:33.008803Z","shell.execute_reply":"2023-08-03T12:13:58.172458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic exploration of the prepared dataset","metadata":{"id":"Ij7TT3x-n2PB"}},{"cell_type":"markdown","source":"Let us print out the first 5 entries using the following code:","metadata":{"id":"s1c59fMAn2PB"}},{"cell_type":"code","source":"dataset_df.head(5)","metadata":{"id":"mvQEsdV1n2PB","execution":{"iopub.status.busy":"2023-08-03T12:13:58.175453Z","iopub.execute_input":"2023-08-03T12:13:58.175995Z","iopub.status.idle":"2023-08-03T12:13:58.204509Z","shell.execute_reply.started":"2023-08-03T12:13:58.175954Z","shell.execute_reply":"2023-08-03T12:13:58.203200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the dataset into training and testing datasets","metadata":{"id":"W0ex_Jkln2PC"}},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"id":"OZfTcCJfn2PC","execution":{"iopub.status.busy":"2023-08-03T12:13:58.205874Z","iopub.execute_input":"2023-08-03T12:13:58.206723Z","iopub.status.idle":"2023-08-03T12:13:58.327912Z","shell.execute_reply.started":"2023-08-03T12:13:58.206673Z","shell.execute_reply":"2023-08-03T12:13:58.326403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create the gradient boosted trees model","metadata":{"id":"UdibIrM-XP5-"}},{"cell_type":"markdown","source":"## Train a model for each question","metadata":{"id":"BtkKMa7sXd65"}},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nVALID_USER_LIST = valid_x.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}","metadata":{"id":"7Brds67Wn2PD","execution":{"iopub.status.busy":"2023-08-03T12:13:58.329806Z","iopub.execute_input":"2023-08-03T12:13:58.330204Z","iopub.status.idle":"2023-08-03T12:13:58.343171Z","shell.execute_reply.started":"2023-08-03T12:13:58.330165Z","shell.execute_reply":"2023-08-03T12:13:58.341440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n        \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = train_x.loc[train_x.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp]\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # There's one more step required before we can train the model. \n    # We need to convert the datatset from Pandas format (pd.DataFrame)\n    # into TensorFlow Datasets format (tf.data.Dataset).\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n    \n    # We will now create the Gradient Boosted Trees Model with default settings. \n    gbtm = tfdf.keras.GradientBoostedTreesModel(verbose=0\n                                                # , num_trees=40\n                                                , max_depth=4\n                                                , shrinkage = 0.02\n                                                , num_candidate_attributes = 0\n                                               )\n    gbtm.compile(metrics=[\"accuracy\"])\n\n    # Train the model\n    gbtm.fit(x=train_ds)\n    \n    # Store the model\n    models[f'{grp}_{q_no}'] = gbtm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    inspector = gbtm.make_inspector()\n    inspector.evaluation()\n    evaluation = gbtm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]\n    # for name, value in evaluation.items():\n    #     print(f\"{name}: {value:.4f}\")\n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = gbtm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()","metadata":{"id":"p1GoOMRHn2PF","execution":{"iopub.status.busy":"2023-08-03T12:13:58.345650Z","iopub.execute_input":"2023-08-03T12:13:58.346798Z","iopub.status.idle":"2023-08-03T12:18:48.760292Z","shell.execute_reply.started":"2023-08-03T12:13:58.346741Z","shell.execute_reply":"2023-08-03T12:18:48.758935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate the accuracy of the models on validation dataset","metadata":{"id":"oOdr1p-Bn2PF"}},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"id":"qPOfPkm7n2PG","execution":{"iopub.status.busy":"2023-08-03T12:18:48.762327Z","iopub.execute_input":"2023-08-03T12:18:48.762744Z","iopub.status.idle":"2023-08-03T12:18:48.771507Z","shell.execute_reply.started":"2023-08-03T12:18:48.762703Z","shell.execute_reply":"2023-08-03T12:18:48.770261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot the training logs","metadata":{}},{"cell_type":"code","source":"models_logs = [models[group].make_inspector().training_logs() for group in models]\n\nfor logs, group in zip(models_logs, models):\n    plt.figure(figsize=(12, 4))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot([log.num_trees for log in logs], [log.evaluation.accuracy for log in logs])\n    \n    plt.xlabel(\"Number of trees\")\n    plt.ylabel(\"Accuracy (out-of-bag)\")\n    plt.title(f\"{group} accuracy vs no. of trees\")\n    \n    plt.subplot(1, 2, 2)\n    plt.plot([log.num_trees for log in logs], [log.evaluation.loss for log in logs])\n    plt.xlabel(\"Number of trees\")\n    plt.ylabel(\"Logloss (out-of-bag)\")\n    plt.title(f\"{group} log loss vs no. of trees\")\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T12:18:48.776640Z","iopub.execute_input":"2023-08-03T12:18:48.777596Z","iopub.status.idle":"2023-08-03T12:18:55.579865Z","shell.execute_reply.started":"2023-08-03T12:18:48.777543Z","shell.execute_reply":"2023-08-03T12:18:55.578670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot the trees","metadata":{"id":"qtiCtfNCn2PG"}},{"cell_type":"code","source":"for group in models:\n    display(tfdf.model_plotter.plot_model_in_colab(models[group], tree_idx=0, max_depth=3))","metadata":{"id":"od__6uAan2PG","execution":{"iopub.status.busy":"2023-08-03T12:18:55.581445Z","iopub.execute_input":"2023-08-03T12:18:55.582545Z","iopub.status.idle":"2023-08-03T12:18:55.713224Z","shell.execute_reply.started":"2023-08-03T12:18:55.582481Z","shell.execute_reply":"2023-08-03T12:18:55.711786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find the best threshold with highest F1 score","metadata":{"id":"PIK5aUH-n2PH"}},{"cell_type":"code","source":"# Create a dataframe of required size:\n# (no: of users in validation set x no: of questions) initialized to zero values\n# to store true values of the label `correct`. \ntrue_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    # Get the true labels.\n    tmp = labels.loc[labels.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0\n\n# Loop through threshold values from 0.4 to 0.8 and select the threshold with \n# the highest `F1 score`.\nfor threshold in np.arange(0.4,0.8,0.01):\n    metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    if f1_score > max_score:\n        max_score = f1_score\n        best_threshold = threshold\n    \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"id":"2wptRs3In2PH","execution":{"iopub.status.busy":"2023-08-03T12:18:55.715037Z","iopub.execute_input":"2023-08-03T12:18:55.715563Z","iopub.status.idle":"2023-08-03T12:18:57.118773Z","shell.execute_reply.started":"2023-08-03T12:18:55.715520Z","shell.execute_reply":"2023-08-03T12:18:57.116223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"id":"ezA40GQ4n2PH"}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    test_df = pre_processing(test_df)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        test_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, test_df.columns != 'level_group'])\n        predictions = gbtm.predict(x=test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"id":"gHiXTnTVn2PI","execution":{"iopub.status.busy":"2023-08-03T12:18:57.120628Z","iopub.execute_input":"2023-08-03T12:18:57.121002Z","iopub.status.idle":"2023-08-03T12:19:10.524683Z","shell.execute_reply.started":"2023-08-03T12:18:57.120966Z","shell.execute_reply":"2023-08-03T12:19:10.523529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"id":"iYBXokAyn2PI","execution":{"iopub.status.busy":"2023-08-03T12:19:10.526309Z","iopub.execute_input":"2023-08-03T12:19:10.527222Z","iopub.status.idle":"2023-08-03T12:19:11.731431Z","shell.execute_reply.started":"2023-08-03T12:19:10.527178Z","shell.execute_reply":"2023-08-03T12:19:11.729602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References\n\nhttps://www.kaggle.com/code/gusthema/student-performance-w-tensorflow-decision-forests/notebook\n\nhttps://www.tensorflow.org/decision_forests/api_docs/python/tfdf/keras/GradientBoostedTreesModel","metadata":{}}]}