{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30513,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-07T12:25:15.197366Z","iopub.execute_input":"2023-07-07T12:25:15.197848Z","iopub.status.idle":"2023-07-07T12:25:15.216775Z","shell.execute_reply.started":"2023-07-07T12:25:15.197816Z","shell.execute_reply":"2023-07-07T12:25:15.215383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The objective of the competition is to build a predictive model that can forecast student performance in real-time during game-based learning. The competition likely provides participants with a dataset containing game logs, which records various interactions and activities of students while using educational games. ","metadata":{}},{"cell_type":"markdown","source":"To begin, we need to import the necessary libraries that will be used in our data analysis and modeling tasks. These libraries provide essential functionality for working with data, performing computations, and building machine learning models.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nfrom sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:25:18.996351Z","iopub.execute_input":"2023-07-07T12:25:18.996745Z","iopub.status.idle":"2023-07-07T12:25:27.363321Z","shell.execute_reply.started":"2023-07-07T12:25:18.996715Z","shell.execute_reply":"2023-07-07T12:25:27.362369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TensorFlow Decision Forests v\" + tfdf.__version__)\nprint(\"TensorFlow Addons v\" + tfa.__version__)\nprint(\"TensorFlow v\" + tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:25:30.627284Z","iopub.execute_input":"2023-07-07T12:25:30.628275Z","iopub.status.idle":"2023-07-07T12:25:30.635835Z","shell.execute_reply.started":"2023-07-07T12:25:30.628232Z","shell.execute_reply":"2023-07-07T12:25:30.634367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To address memory-related issues when loading large datasets with Pandas, we can optimize memory usage by considering the following techniques:\n\nBy default, when Pandas loads a dataset, it assigns data types based on its automatic detection. This often results in assigning larger data types such as int64 for numerical columns, float64 for float columns, and object dtype for string columns, regardless of whether the maximum values in these columns require such large types for storage.\n\nTo reduce memory consumption, we can downcast numerical columns to smaller types like int8, int32, float32, etc., as long as the maximum values in these columns do not exceed the capacity of the smaller types (e.g., int64, float64).\n\nAdditionally, Pandas automatically assigns the object datatype to string columns. However, if these string columns store categorical data, we can significantly reduce memory usage by explicitly specifying their data type as \"category\".\n\nBy employing these techniques, we can optimize memory usage when loading and storing datasets with Pandas, which can help mitigate memory errors when dealing with large datasets.","metadata":{}},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:25:35.759530Z","iopub.execute_input":"2023-07-07T12:25:35.759964Z","iopub.status.idle":"2023-07-07T12:27:13.022206Z","shell.execute_reply.started":"2023-07-07T12:25:35.759932Z","shell.execute_reply":"2023-07-07T12:27:13.021318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:13.024050Z","iopub.execute_input":"2023-07-07T12:27:13.024607Z","iopub.status.idle":"2023-07-07T12:27:13.064191Z","shell.execute_reply.started":"2023-07-07T12:27:13.024576Z","shell.execute_reply":"2023-07-07T12:27:13.063320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load label data ","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:13.065487Z","iopub.execute_input":"2023-07-07T12:27:13.065783Z","iopub.status.idle":"2023-07-07T12:27:14.532370Z","shell.execute_reply.started":"2023-07-07T12:27:13.065758Z","shell.execute_reply":"2023-07-07T12:27:14.531472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:14.535172Z","iopub.execute_input":"2023-07-07T12:27:14.535868Z","iopub.status.idle":"2023-07-07T12:27:14.547476Z","shell.execute_reply.started":"2023-07-07T12:27:14.535833Z","shell.execute_reply":"2023-07-07T12:27:14.546315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:14.560378Z","iopub.execute_input":"2023-07-07T12:27:14.560926Z","iopub.status.idle":"2023-07-07T12:27:14.574757Z","shell.execute_reply.started":"2023-07-07T12:27:14.560880Z","shell.execute_reply":"2023-07-07T12:27:14.573539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:14.576390Z","iopub.execute_input":"2023-07-07T12:27:14.576759Z","iopub.status.idle":"2023-07-07T12:27:14.589436Z","shell.execute_reply.started":"2023-07-07T12:27:14.576730Z","shell.execute_reply":"2023-07-07T12:27:14.588218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = feature_engineer(dataset_df)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:14.591366Z","iopub.execute_input":"2023-07-07T12:27:14.591725Z","iopub.status.idle":"2023-07-07T12:27:55.595269Z","shell.execute_reply.started":"2023-07-07T12:27:14.591695Z","shell.execute_reply":"2023-07-07T12:27:55.594424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Data visualization ","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:55.596354Z","iopub.execute_input":"2023-07-07T12:27:55.597227Z","iopub.status.idle":"2023-07-07T12:27:55.601006Z","shell.execute_reply.started":"2023-07-07T12:27:55.597197Z","shell.execute_reply":"2023-07-07T12:27:55.600221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:55.604397Z","iopub.execute_input":"2023-07-07T12:27:55.605190Z","iopub.status.idle":"2023-07-07T12:27:55.641369Z","shell.execute_reply.started":"2023-07-07T12:27:55.605146Z","shell.execute_reply":"2023-07-07T12:27:55.640324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.info","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:55.642448Z","iopub.execute_input":"2023-07-07T12:27:55.643233Z","iopub.status.idle":"2023-07-07T12:27:55.670793Z","shell.execute_reply.started":"2023-07-07T12:27:55.643203Z","shell.execute_reply":"2023-07-07T12:27:55.669675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:55.672315Z","iopub.execute_input":"2023-07-07T12:27:55.672649Z","iopub.status.idle":"2023-07-07T12:27:55.813448Z","shell.execute_reply.started":"2023-07-07T12:27:55.672622Z","shell.execute_reply":"2023-07-07T12:27:55.812590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = dataset_df.corr()\n# Visualize the correlation matrix\nplt.figure(figsize=(18, 12))  # Increase the figure size\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix', fontsize=16)  # Increase the title font size\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:55.814661Z","iopub.execute_input":"2023-07-07T12:27:55.815180Z","iopub.status.idle":"2023-07-07T12:27:57.665609Z","shell.execute_reply.started":"2023-07-07T12:27:55.815141Z","shell.execute_reply":"2023-07-07T12:27:57.664724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.hist(bins=20, figsize=(12, 10))\nplt.tight_layout()\nplt.show()\n\n# Plot box plots of selected columns\nplt.figure(figsize=(12, 10))\nsns.boxplot(data=dataset_df)\nplt.xticks(rotation=90)\nplt.title('Box Plots of Columns')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:27:57.666566Z","iopub.execute_input":"2023-07-07T12:27:57.666849Z","iopub.status.idle":"2023-07-07T12:28:03.026755Z","shell.execute_reply.started":"2023-07-07T12:27:57.666823Z","shell.execute_reply":"2023-07-07T12:28:03.025934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df['elapsed_time'] = pd.to_datetime(dataset_df['elapsed_time'], unit='ms')\ndataset_df['hour'] = dataset_df['elapsed_time'].dt.hour\ndataset_df['day'] = dataset_df['elapsed_time'].dt.day\ndataset_df['month'] = dataset_df['elapsed_time'].dt.month\ndataset_df['year'] = dataset_df['elapsed_time'].dt.year","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.028155Z","iopub.execute_input":"2023-07-07T12:28:03.028736Z","iopub.status.idle":"2023-07-07T12:28:03.075271Z","shell.execute_reply.started":"2023-07-07T12:28:03.028704Z","shell.execute_reply":"2023-07-07T12:28:03.074385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.drop('elapsed_time',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.076714Z","iopub.execute_input":"2023-07-07T12:28:03.077330Z","iopub.status.idle":"2023-07-07T12:28:03.086209Z","shell.execute_reply.started":"2023-07-07T12:28:03.077299Z","shell.execute_reply":"2023-07-07T12:28:03.085318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.087671Z","iopub.execute_input":"2023-07-07T12:28:03.088380Z","iopub.status.idle":"2023-07-07T12:28:03.198091Z","shell.execute_reply.started":"2023-07-07T12:28:03.088349Z","shell.execute_reply":"2023-07-07T12:28:03.197284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.keras.get_all_models()","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.199470Z","iopub.execute_input":"2023-07-07T12:28:03.200094Z","iopub.status.idle":"2023-07-07T12:28:03.205680Z","shell.execute_reply.started":"2023-07-07T12:28:03.200056Z","shell.execute_reply":"2023-07-07T12:28:03.204955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = tfdf.keras.GradientBoostedTreesModel(hyperparameter_template=\"benchmark_rank1\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.206845Z","iopub.execute_input":"2023-07-07T12:28:03.207375Z","iopub.status.idle":"2023-07-07T12:28:03.490216Z","shell.execute_reply.started":"2023-07-07T12:28:03.207346Z","shell.execute_reply":"2023-07-07T12:28:03.488112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nVALID_USER_LIST = valid_x.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.491403Z","iopub.execute_input":"2023-07-07T12:28:03.491696Z","iopub.status.idle":"2023-07-07T12:28:03.501460Z","shell.execute_reply.started":"2023-07-07T12:28:03.491669Z","shell.execute_reply":"2023-07-07T12:28:03.500565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configure the tuner.\n\n# Create a Random Search tuner with 50 trials.\ntuner = tfdf.tuner.RandomSearch(num_trials=50)\n\n# Define the search space.\n#\n# Adding more parameters generaly improve the quality of the model, but make\n# the tuning last longer.\n\ntuner.choice(\"min_examples\", [2, 5, 7, 10])\ntuner.choice(\"categorical_algorithm\", [\"CART\", \"RANDOM\"])\n\n# Some hyper-parameters are only valid for specific values of other\n# hyper-parameters. For example, the \"max_depth\" parameter is mostly useful when\n# \"growing_strategy=LOCAL\" while \"max_num_nodes\" is better suited when\n# \"growing_strategy=BEST_FIRST_GLOBAL\".\n\nlocal_search_space = tuner.choice(\"growing_strategy\", [\"LOCAL\"])\nlocal_search_space.choice(\"max_depth\", [3, 4, 5, 6, 8])\n\n# merge=True indicates that the parameter (here \"growing_strategy\") is already\n# defined, and that new values are added to it.\nglobal_search_space = tuner.choice(\"growing_strategy\", [\"BEST_FIRST_GLOBAL\"], merge=True)\nglobal_search_space.choice(\"max_num_nodes\", [16, 32, 64, 128, 256])\n\ntuner.choice(\"use_hessian_gain\", [True, False])\ntuner.choice(\"shrinkage\", [0.02, 0.05, 0.10, 0.15])\ntuner.choice(\"num_candidate_attributes_ratio\", [0.2, 0.5, 0.9, 1.0])\n\n# Uncomment some (or all) of the following hyper-parameters to increase the\n# quality of the search. The number of trial should be increased accordingly.\n\n# tuner.choice(\"split_axis\", [\"AXIS_ALIGNED\"])\n# oblique_space = tuner.choice(\"split_axis\", [\"SPARSE_OBLIQUE\"], merge=True)\n# oblique_space.choice(\"sparse_oblique_normalization\",\n#                      [\"NONE\", \"STANDARD_DEVIATION\", \"MIN_MAX\"])\n# oblique_space.choice(\"sparse_oblique_weights\", [\"BINARY\", \"CONTINUOUS\"])\n# oblique_space.choice(\"sparse_oblique_num_projections_exponent\", [1.0, 1.5])","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.502940Z","iopub.execute_input":"2023-07-07T12:28:03.503528Z","iopub.status.idle":"2023-07-07T12:28:03.518164Z","shell.execute_reply.started":"2023-07-07T12:28:03.503498Z","shell.execute_reply":"2023-07-07T12:28:03.516949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n        \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = train_x.loc[train_x.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp]\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # There's one more step required before we can train the model. \n    # We need to convert the datatset from Pandas format (pd.DataFrame)\n    # into TensorFlow Datasets format (tf.data.Dataset).\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n\n    # We will now create the Gradient Boosted Trees Model with default settings. \n    # By default the model is set to train for a classification task.\n    gbtm = tfdf.keras.RandomForestModel()\n    gbtm.compile(metrics=[\"accuracy\"]) \n\n    # Train the model.\n    gbtm.fit(x=train_ds)\n\n    # Store the model\n    models[f'{grp}_{q_no}'] = gbtm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    inspector = gbtm.make_inspector()\n    inspector.evaluation()\n    evaluation = gbtm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]         \n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = gbtm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()     ","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:28:03.519791Z","iopub.execute_input":"2023-07-07T12:28:03.520514Z","iopub.status.idle":"2023-07-07T12:35:10.197325Z","shell.execute_reply.started":"2023-07-07T12:28:03.520483Z","shell.execute_reply":"2023-07-07T12:35:10.196457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n    print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:36:23.920607Z","iopub.execute_input":"2023-07-07T12:36:23.921177Z","iopub.status.idle":"2023-07-07T12:36:23.929465Z","shell.execute_reply.started":"2023-07-07T12:36:23.921139Z","shell.execute_reply":"2023-07-07T12:36:23.928503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.model_plotter.plot_model_in_colab(models['0-4_1'], tree_idx=0, max_depth=3)\ninspector = models['0-4_1'].make_inspector()\n\nprint(f\"Available variable importances:\")\nfor importance in inspector.variable_importances().keys():\n    print(\"\\t\", importance)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:40:43.230548Z","iopub.execute_input":"2023-07-07T12:40:43.231185Z","iopub.status.idle":"2023-07-07T12:40:43.309505Z","shell.execute_reply.started":"2023-07-07T12:40:43.231139Z","shell.execute_reply":"2023-07-07T12:40:43.308632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Each line is: (feature name, (index of the feature), importance score)\ninspector.variable_importances()[\"NUM_AS_ROOT\"]","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:40:58.786712Z","iopub.execute_input":"2023-07-07T12:40:58.787180Z","iopub.status.idle":"2023-07-07T12:40:58.797493Z","shell.execute_reply.started":"2023-07-07T12:40:58.787148Z","shell.execute_reply":"2023-07-07T12:40:58.795969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe of required size:\n# (no: of users in validation set x no: of questions) initialized to zero values\n# to store true values of the label `correct`. \ntrue_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    # Get the true labels.\n    tmp = labels.loc[labels.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0\n\n# Loop through threshold values from 0.4 to 0.8 and select the threshold with \n# the highest `F1 score`.\nfor threshold in np.arange(0.4,0.8,0.01):\n    metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    if f1_score > max_score:\n        max_score = f1_score\n        best_threshold = threshold\n        \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T12:41:40.308026Z","iopub.execute_input":"2023-07-07T12:41:40.308561Z","iopub.status.idle":"2023-07-07T12:41:41.575745Z","shell.execute_reply.started":"2023-07-07T12:41:40.308526Z","shell.execute_reply":"2023-07-07T12:41:41.574864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}