{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T07:09:30.080666Z","iopub.execute_input":"2023-06-08T07:09:30.081489Z","iopub.status.idle":"2023-06-08T07:09:30.128700Z","shell.execute_reply.started":"2023-06-08T07:09:30.081449Z","shell.execute_reply":"2023-06-08T07:09:30.127726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:29:00.936784Z","iopub.execute_input":"2023-06-08T07:29:00.937236Z","iopub.status.idle":"2023-06-08T07:29:11.053570Z","shell.execute_reply.started":"2023-06-08T07:29:00.937197Z","shell.execute_reply":"2023-06-08T07:29:11.052362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:31:45.813585Z","iopub.execute_input":"2023-06-08T07:31:45.815518Z","iopub.status.idle":"2023-06-08T07:34:04.247526Z","shell.execute_reply.started":"2023-06-08T07:31:45.815467Z","shell.execute_reply":"2023-06-08T07:34:04.246234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:35:00.353989Z","iopub.execute_input":"2023-06-08T07:35:00.354438Z","iopub.status.idle":"2023-06-08T07:35:00.406483Z","shell.execute_reply.started":"2023-06-08T07:35:00.354405Z","shell.execute_reply":"2023-06-08T07:35:00.405462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:39:12.929080Z","iopub.execute_input":"2023-06-08T07:39:12.929716Z","iopub.status.idle":"2023-06-08T07:39:13.274845Z","shell.execute_reply.started":"2023-06-08T07:39:12.929673Z","shell.execute_reply":"2023-06-08T07:39:13.273800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nlabels.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:39:18.385883Z","iopub.execute_input":"2023-06-08T07:39:18.386510Z","iopub.status.idle":"2023-06-08T07:39:19.473687Z","shell.execute_reply.started":"2023-06-08T07:39:18.386478Z","shell.execute_reply":"2023-06-08T07:39:19.472781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(3, 3))\nplot_df = labels.correct.value_counts()\nplot_df.plot(kind=\"bar\", color=['b', 'c'])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:39:45.880032Z","iopub.execute_input":"2023-06-08T07:39:45.880507Z","iopub.status.idle":"2023-06-08T07:39:46.171444Z","shell.execute_reply.started":"2023-06-08T07:39:45.880477Z","shell.execute_reply":"2023-06-08T07:39:46.169802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"\\\"Correct\\\" column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    ax = plt.subplot(6, 3, n)\n\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['b', 'c'])\n    \n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:40:42.912124Z","iopub.execute_input":"2023-06-08T07:40:42.912577Z","iopub.status.idle":"2023-06-08T07:40:45.991690Z","shell.execute_reply.started":"2023-06-08T07:40:42.912544Z","shell.execute_reply":"2023-06-08T07:40:45.990702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:42:48.859833Z","iopub.execute_input":"2023-06-08T07:42:48.860354Z","iopub.status.idle":"2023-06-08T07:42:48.866760Z","shell.execute_reply.started":"2023-06-08T07:42:48.860318Z","shell.execute_reply":"2023-06-08T07:42:48.865519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:43:35.797985Z","iopub.execute_input":"2023-06-08T07:43:35.798557Z","iopub.status.idle":"2023-06-08T07:43:35.810116Z","shell.execute_reply.started":"2023-06-08T07:43:35.798515Z","shell.execute_reply":"2023-06-08T07:43:35.808841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = feature_engineer(dataset_df)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:43:48.077506Z","iopub.execute_input":"2023-06-08T07:43:48.077955Z","iopub.status.idle":"2023-06-08T07:44:29.826258Z","shell.execute_reply.started":"2023-06-08T07:43:48.077926Z","shell.execute_reply":"2023-06-08T07:44:29.825299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:46:56.983642Z","iopub.execute_input":"2023-06-08T07:46:56.984167Z","iopub.status.idle":"2023-06-08T07:46:57.018024Z","shell.execute_reply.started":"2023-06-08T07:46:56.984134Z","shell.execute_reply":"2023-06-08T07:46:57.017038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:47:18.459864Z","iopub.execute_input":"2023-06-08T07:47:18.460921Z","iopub.status.idle":"2023-06-08T07:47:18.618233Z","shell.execute_reply.started":"2023-06-08T07:47:18.460879Z","shell.execute_reply":"2023-06-08T07:47:18.616880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figure, axis = plt.subplots(3, 2, figsize=(10, 10))\n\nfor name, data in dataset_df.groupby('level_group'):\n    axis[0, 0].plot(range(1, len(data['room_coor_x_std'])+1), data['room_coor_x_std'], label=name)\n    axis[0, 1].plot(range(1, len(data['room_coor_y_std'])+1), data['room_coor_y_std'], label=name)\n    axis[1, 0].plot(range(1, len(data['screen_coor_x_std'])+1), data['screen_coor_x_std'], label=name)\n    axis[1, 1].plot(range(1, len(data['screen_coor_y_std'])+1), data['screen_coor_y_std'], label=name)\n    axis[2, 0].plot(range(1, len(data['hover_duration'])+1), data['hover_duration_std'], label=name)\n    axis[2, 1].plot(range(1, len(data['elapsed_time_std'])+1), data['elapsed_time_std'], label=name)\n    \n\naxis[0, 0].set_title('room_coor_x')\naxis[0, 1].set_title('room_coor_y')\naxis[1, 0].set_title('screen_coor_x')\naxis[1, 1].set_title('screen_coor_y')\naxis[2, 0].set_title('hover_duration')\naxis[2, 1].set_title('elapsed_time_std')\n\nfor i in range(3):\n    axis[i, 0].legend()\n    axis[i, 1].legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:49:13.686601Z","iopub.execute_input":"2023-06-08T07:49:13.687097Z","iopub.status.idle":"2023-06-08T07:49:17.265231Z","shell.execute_reply.started":"2023-06-08T07:49:13.687066Z","shell.execute_reply":"2023-06-08T07:49:17.264294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:50:18.439479Z","iopub.execute_input":"2023-06-08T07:50:18.440000Z","iopub.status.idle":"2023-06-08T07:50:18.544624Z","shell.execute_reply.started":"2023-06-08T07:50:18.439965Z","shell.execute_reply":"2023-06-08T07:50:18.543387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.keras.get_all_models()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:50:44.564695Z","iopub.execute_input":"2023-06-08T07:50:44.565155Z","iopub.status.idle":"2023-06-08T07:50:44.573569Z","shell.execute_reply.started":"2023-06-08T07:50:44.565124Z","shell.execute_reply":"2023-06-08T07:50:44.572279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VALID_USER_LIST = valid_x.index.unique()\nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nmodels = {}\nevaluation_dict ={}","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:52:43.825457Z","iopub.execute_input":"2023-06-08T07:52:43.825995Z","iopub.status.idle":"2023-06-08T07:52:43.834472Z","shell.execute_reply.started":"2023-06-08T07:52:43.825962Z","shell.execute_reply":"2023-06-08T07:52:43.833146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for q_no in range(1,19):\n\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n    train_df = train_x.loc[train_x.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp]\n    valid_users = valid_df.index.values\n\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n\n    gbtm = tfdf.keras.GradientBoostedTreesModel(verbose=0)\n    gbtm.compile(metrics=[\"accuracy\"])\n\n    gbtm.fit(x=train_ds)\n\n    models[f'{grp}_{q_no}'] = gbtm\n\n    inspector = gbtm.make_inspector()\n    inspector.evaluation()\n    evaluation = gbtm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]         \n\n    predict = gbtm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:54:26.472572Z","iopub.execute_input":"2023-06-08T07:54:26.473065Z","iopub.status.idle":"2023-06-08T07:57:05.063382Z","shell.execute_reply.started":"2023-06-08T07:54:26.473032Z","shell.execute_reply":"2023-06-08T07:57:05.062308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:57:34.021109Z","iopub.execute_input":"2023-06-08T07:57:34.022035Z","iopub.status.idle":"2023-06-08T07:57:34.029162Z","shell.execute_reply.started":"2023-06-08T07:57:34.021995Z","shell.execute_reply":"2023-06-08T07:57:34.027894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfdf.model_plotter.plot_model_in_colab(models['0-4_1'], tree_idx=0, max_depth=3)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:57:53.148949Z","iopub.execute_input":"2023-06-08T07:57:53.149373Z","iopub.status.idle":"2023-06-08T07:57:53.163996Z","shell.execute_reply.started":"2023-06-08T07:57:53.149343Z","shell.execute_reply":"2023-06-08T07:57:53.162864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inspector = models['0-4_1'].make_inspector()\n\nprint(f\"Available variable importances:\")\nfor importance in inspector.variable_importances().keys():\n    print(\"\\t\", importance)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:58:30.061281Z","iopub.execute_input":"2023-06-08T07:58:30.062409Z","iopub.status.idle":"2023-06-08T07:58:30.072066Z","shell.execute_reply.started":"2023-06-08T07:58:30.062369Z","shell.execute_reply":"2023-06-08T07:58:30.071000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inspector.variable_importances()[\"NUM_AS_ROOT\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:58:46.320741Z","iopub.execute_input":"2023-06-08T07:58:46.321418Z","iopub.status.idle":"2023-06-08T07:58:46.328623Z","shell.execute_reply.started":"2023-06-08T07:58:46.321384Z","shell.execute_reply":"2023-06-08T07:58:46.327573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    tmp = labels.loc[labels.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.8,0.01):\n    metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    if f1_score > max_score:\n        max_score = f1_score\n        best_threshold = threshold\n        \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T07:59:49.621460Z","iopub.execute_input":"2023-06-08T07:59:49.621881Z","iopub.status.idle":"2023-06-08T07:59:50.496675Z","shell.execute_reply.started":"2023-06-08T07:59:49.621849Z","shell.execute_reply":"2023-06-08T07:59:50.494045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\nimport jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        test_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, test_df.columns != 'level_group'])\n        predictions = gbtm.predict(test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:01:29.670380Z","iopub.execute_input":"2023-06-08T08:01:29.670837Z","iopub.status.idle":"2023-06-08T08:01:35.721531Z","shell.execute_reply.started":"2023-06-08T08:01:29.670804Z","shell.execute_reply":"2023-06-08T08:01:35.720243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-08T08:01:44.036702Z","iopub.execute_input":"2023-06-08T08:01:44.037146Z","iopub.status.idle":"2023-06-08T08:01:45.264016Z","shell.execute_reply.started":"2023-06-08T08:01:44.037113Z","shell.execute_reply":"2023-06-08T08:01:45.262567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}