{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:24:19.758551Z","iopub.execute_input":"2023-05-19T02:24:19.758949Z","iopub.status.idle":"2023-05-19T02:24:19.764497Z","shell.execute_reply.started":"2023-05-19T02:24:19.758919Z","shell.execute_reply":"2023-05-19T02:24:19.763474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndf = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:24:19.773636Z","iopub.execute_input":"2023-05-19T02:24:19.774326Z","iopub.status.idle":"2023-05-19T02:26:05.860785Z","shell.execute_reply.started":"2023-05-19T02:24:19.774280Z","shell.execute_reply":"2023-05-19T02:26:05.859903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load labels dataset\nlabels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:05.862728Z","iopub.execute_input":"2023-05-19T02:26:05.863251Z","iopub.status.idle":"2023-05-19T02:26:06.229588Z","shell.execute_reply.started":"2023-05-19T02:26:05.863208Z","shell.execute_reply":"2023-05-19T02:26:06.228406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the session_id with the question column\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:06.231136Z","iopub.execute_input":"2023-05-19T02:26:06.231468Z","iopub.status.idle":"2023-05-19T02:26:06.908186Z","shell.execute_reply.started":"2023-05-19T02:26:06.231440Z","shell.execute_reply":"2023-05-19T02:26:06.906898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:06.909918Z","iopub.execute_input":"2023-05-19T02:26:06.910538Z","iopub.status.idle":"2023-05-19T02:26:06.915412Z","shell.execute_reply.started":"2023-05-19T02:26:06.910503Z","shell.execute_reply":"2023-05-19T02:26:06.914384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For each categorical column, we will first group the dataset by session_id and level_group. We will then count the number of distinct elements in the column for each group and store it temporarily.\n\nFor all **numerical columns**, we will group the dataset by **session id** and **level_group**. Instead of counting the number of distinct elements, we will calculate the **mean and **standard deviation** of the numerical column for each group and store it temporarily.\n\nAfter this, we will concatenate the temporary data frames we generated in the earlier step for each column to create our new feature engineered dataset.**","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\ndef feature_engineer(df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:06.918484Z","iopub.execute_input":"2023-05-19T02:26:06.918863Z","iopub.status.idle":"2023-05-19T02:26:06.931543Z","shell.execute_reply.started":"2023-05-19T02:26:06.918822Z","shell.execute_reply":"2023-05-19T02:26:06.930406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = feature_engineer(df)\nprint(\"Full prepared dataset shape is {}\".format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:06.933110Z","iopub.execute_input":"2023-05-19T02:26:06.933649Z","iopub.status.idle":"2023-05-19T02:26:45.044824Z","shell.execute_reply.started":"2023-05-19T02:26:06.933615Z","shell.execute_reply":"2023-05-19T02:26:45.043919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check missing values if any\nmissing_values = df.isnull().sum()\nmissing_values","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.045940Z","iopub.execute_input":"2023-05-19T02:26:45.046448Z","iopub.status.idle":"2023-05-19T02:26:45.059433Z","shell.execute_reply.started":"2023-05-19T02:26:45.046417Z","shell.execute_reply":"2023-05-19T02:26:45.058396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.061066Z","iopub.execute_input":"2023-05-19T02:26:45.061648Z","iopub.status.idle":"2023-05-19T02:26:45.086770Z","shell.execute_reply.started":"2023-05-19T02:26:45.061473Z","shell.execute_reply":"2023-05-19T02:26:45.085749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def split_dataset(dataset, test_ratio=0.20):\n#     USER_LIST = dataset.index.unique()\n#     split = int(len(USER_LIST) * (1 - 0.20))\n#     return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\n# train_x, valid_x = split_dataset(dataset_df)\n# print(\"{} examples in training, {} examples in testing.\".format(\n#     len(train_x), len(valid_x)))\n# 56547 examples in training, 14139 examples in testing.","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.088043Z","iopub.execute_input":"2023-05-19T02:26:45.088765Z","iopub.status.idle":"2023-05-19T02:26:45.096005Z","shell.execute_reply.started":"2023-05-19T02:26:45.088732Z","shell.execute_reply":"2023-05-19T02:26:45.094928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Error in this part \n#  split the data\ndef split_data(data, test_ratio=0.20):\n    user_list = data.index.unique()\n    split = int(len(user_list) * (1 - 0.20))\n    return data.loc[user_list[:split]], data.loc[user_list[split:]]\n\nX_train, X_valid = split_data(df)\nX_train.shape, X_valid.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.097450Z","iopub.execute_input":"2023-05-19T02:26:45.097876Z","iopub.status.idle":"2023-05-19T02:26:45.170985Z","shell.execute_reply.started":"2023-05-19T02:26:45.097836Z","shell.execute_reply":"2023-05-19T02:26:45.169966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Define hyperparameters for each model\nmodel_hyperparameters = [\n    # Model 1: Logistic Regression\n    {'C': 1.0, 'penalty': 'l2'},\n    \n    # Model 2: Support Vector Machine\n    {'C': 1.0, 'kernel': 'rbf'},\n    \n    # Model 3: Random Forest\n    {'n_estimators': 100, 'max_depth': 10},\n    \n    # Model 4: Gradient Boosting Machine\n    {'n_estimators': 100, 'learning_rate': 0.1},\n    \n    # Model 5: Naive Bayes\n    {},\n    \n    # Model 6: Neural Network\n    {'hidden_layer_sizes': (100,)}\n]\n\n# Create a list of models\nmodels = [\n    LogisticRegression(),\n    SVC(),\n    RandomForestClassifier(),\n    GradientBoostingClassifier(),\n    GaussianNB(),\n    MLPClassifier()\n]\n\n# Train and evaluate each model in a loop\nfor i in range(len(models)):\n    model = models[i]\n    hyperparams = model_hyperparameters[i]\n    \n    # Set the hyperparameters of the model\n    model.set_params(**hyperparams)\n    \n   \n","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.172537Z","iopub.execute_input":"2023-05-19T02:26:45.172898Z","iopub.status.idle":"2023-05-19T02:26:45.181361Z","shell.execute_reply.started":"2023-05-19T02:26:45.172869Z","shell.execute_reply":"2023-05-19T02:26:45.180487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.keras.get_all_models()\nrf = tfdf.keras.GradientBoostedTreesModel(hyperparameter_template=\"benchmark_rank1\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.182637Z","iopub.execute_input":"2023-05-19T02:26:45.183140Z","iopub.status.idle":"2023-05-19T02:26:45.237450Z","shell.execute_reply.started":"2023-05-19T02:26:45.183110Z","shell.execute_reply":"2023-05-19T02:26:45.236423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Fetch the unique list of user sessions in the validation dataset. We assigned \n# # `session_id` as the index of our feature engineered dataset. Hence fetching \n# # the unique values in the index column will give us a list of users in the \n# # validation set.\n# VALID_USER_LIST = valid_x.index.unique()\n\n# # Create a dataframe for storing the predictions of each question for all users\n# # in the validation set.\n# # For this, the required size of the data frame is: \n# # (no: of users in validation set  x no of questions).\n# # We will initialize all the predicted values in the data frame to zero.\n# # The dataframe's index column is the user `session_id`s. \n# prediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# # Create an empty dictionary to store the models created for each question.\n# models = {}\n\n# # Create an empty dictionary to store the evaluation score for each question.\n# evaluation_dict ={}","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.238732Z","iopub.execute_input":"2023-05-19T02:26:45.239043Z","iopub.status.idle":"2023-05-19T02:26:45.244108Z","shell.execute_reply.started":"2023-05-19T02:26:45.239016Z","shell.execute_reply":"2023-05-19T02:26:45.243212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nvalid_users_list = X_valid.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(valid_users_list),18)), index=valid_users_list)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.247351Z","iopub.execute_input":"2023-05-19T02:26:45.247899Z","iopub.status.idle":"2023-05-19T02:26:45.264687Z","shell.execute_reply.started":"2023-05-19T02:26:45.247865Z","shell.execute_reply":"2023-05-19T02:26:45.263628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n        \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = X_train.loc[X_train.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = X_valid.loc[X_valid.level_group == grp]\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # There's one more step required before we can train the model. \n    # We need to convert the datatset from Pandas format (pd.DataFrame)\n    # into TensorFlow Datasets format (tf.data.Dataset).\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n\n    # We will now create the Gradient Boosted Trees Model with default settings. \n    # By default the model is set to train for a classification task.\n    gbtm = tfdf.keras.GradientBoostedTreesModel(verbose=0)\n    gbtm.compile(metrics=[\"accuracy\"])\n\n    # Train the model.\n    gbtm.fit(x=train_ds)\n\n    # Store the model\n    models[f'{grp}_{q_no}'] = gbtm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    inspector = gbtm.make_inspector()\n    inspector.evaluation()\n    evaluation = gbtm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]         \n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = gbtm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()    ","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:26:45.266265Z","iopub.execute_input":"2023-05-19T02:26:45.266962Z","iopub.status.idle":"2023-05-19T02:29:08.214773Z","shell.execute_reply.started":"2023-05-19T02:26:45.266927Z","shell.execute_reply":"2023-05-19T02:29:08.213823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:29:08.216187Z","iopub.execute_input":"2023-05-19T02:29:08.216772Z","iopub.status.idle":"2023-05-19T02:29:08.222319Z","shell.execute_reply.started":"2023-05-19T02:29:08.216740Z","shell.execute_reply":"2023-05-19T02:29:08.221437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.model_plotter.plot_model_in_colab(models['0-4_1'], tree_idx=0, max_depth=3)","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:29:08.223651Z","iopub.execute_input":"2023-05-19T02:29:08.224180Z","iopub.status.idle":"2023-05-19T02:29:08.248151Z","shell.execute_reply.started":"2023-05-19T02:29:08.224149Z","shell.execute_reply":"2023-05-19T02:29:08.247205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inspector = models['0-4_1'].make_inspector()\n\nprint(f\"Available variable importances:\")\nfor importance in inspector.variable_importances().keys():\n  print(\"\\t\", importance)","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:29:08.249825Z","iopub.execute_input":"2023-05-19T02:29:08.250365Z","iopub.status.idle":"2023-05-19T02:29:08.258260Z","shell.execute_reply.started":"2023-05-19T02:29:08.250333Z","shell.execute_reply":"2023-05-19T02:29:08.257328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Each line is: (feature name, (index of the feature), importance score)\ninspector.variable_importances()[\"NUM_AS_ROOT\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:29:08.259587Z","iopub.execute_input":"2023-05-19T02:29:08.260101Z","iopub.status.idle":"2023-05-19T02:29:08.270150Z","shell.execute_reply.started":"2023-05-19T02:29:08.260069Z","shell.execute_reply":"2023-05-19T02:29:08.269193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe of required size:\n# (no: of users in validation set x no: of questions) initialized to zero values\n# to store true values of the label `correct`. \ntrue_df = pd.DataFrame(data=np.zeros((len(valid_users_list),18)), index=valid_users_list)\nfor i in range(18):\n    # Get the true labels.\n    tmp = labels.loc[labels.q == i+1].set_index('session').loc[valid_users_list]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0\n\n# Loop through threshold values from 0.4 to 0.8 and select the threshold with \n# the highest `F1 score`.\nfor threshold in np.arange(0.4,0.8,0.01):\n    metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    if f1_score > max_score:\n        max_score = f1_score\n        best_threshold = threshold\n        \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-05-19T02:40:53.272660Z","iopub.execute_input":"2023-05-19T02:40:53.273165Z","iopub.status.idle":"2023-05-19T02:40:54.470112Z","shell.execute_reply.started":"2023-05-19T02:40:53.273134Z","shell.execute_reply":"2023-05-19T02:40:54.469266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}