{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"},{"sourceId":154729623,"sourceType":"kernelVersion"}],"dockerImageVersionId":30615,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Predict Student Performance from Game Play - Test Inference**\n\n### **Ho Chi Minh City University of Science**\n\n### **Faculty of Information Technology**\n\n#### **Class: 20KHDL**\n\n#### **Lecturers:**\n- **Dr. Nguyễn Tiến Huy**\n- **Mr. Nguyễn Trần Duy Minh**","metadata":{}},{"cell_type":"markdown","source":"# **Special thank**","metadata":{}},{"cell_type":"markdown","source":"- First of all, we would like to thank to [**Jack (Japan)**](https://www.kaggle.com/rsakata) for his notebooks relating to the competition.\n- For further information about [**Jack (Japan)'s**](https://www.kaggle.com/rsakata) works, please refer to [this discussion](https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/420119).","metadata":{}},{"cell_type":"markdown","source":"# **EDA process**","metadata":{}},{"cell_type":"markdown","source":"> For more information about the EDA process, please refer to the original notebook: https://www.kaggle.com/code/goldencheem/hcmus-fit-2023-dynamind","metadata":{}},{"cell_type":"markdown","source":"# **Outline**","metadata":{}},{"cell_type":"markdown","source":"This file first loads the model trained on the [**FE and Train notebook**](https://www.kaggle.com/code/goldencheem/hcmus-fit-2023-dynamind-fe-and-train) and then uses it predict the test data.","metadata":{}},{"cell_type":"markdown","source":"# **Libraries used**","metadata":{}},{"cell_type":"code","source":"import pickle\nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:43.065716Z","iopub.execute_input":"2023-12-14T02:43:43.066036Z","iopub.status.idle":"2023-12-14T02:43:43.393407Z","shell.execute_reply.started":"2023-12-14T02:43:43.066005Z","shell.execute_reply":"2023-12-14T02:43:43.392770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get path of all input files\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-14T02:43:43.394677Z","iopub.execute_input":"2023-12-14T02:43:43.395118Z","iopub.status.idle":"2023-12-14T02:43:43.415821Z","shell.execute_reply.started":"2023-12-14T02:43:43.395086Z","shell.execute_reply":"2023-12-14T02:43:43.415106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get hidden test environment\nimport jo_wilder_310 as jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:43.419232Z","iopub.execute_input":"2023-12-14T02:43:43.421077Z","iopub.status.idle":"2023-12-14T02:43:43.442976Z","shell.execute_reply.started":"2023-12-14T02:43:43.421048Z","shell.execute_reply":"2023-12-14T02:43:43.441705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path of trained models\nMODEL_DIR = \"/kaggle/input/hcmus-fit-2023-dynamind-fe-and-train/\"","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:43.447975Z","iopub.execute_input":"2023-12-14T02:43:43.449869Z","iopub.status.idle":"2023-12-14T02:43:43.454337Z","shell.execute_reply.started":"2023-12-14T02:43:43.449834Z","shell.execute_reply":"2023-12-14T02:43:43.453726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeatureMaker():\n    def __init__(self):\n        self.map_key = None\n        self.valid_keys = None\n        self.list_text_seq = None\n        self.map_text_seq = None\n        self.feature_names = []\n                \n    def prepare(self, df, threshold):\n        # concatenation of variables\n        keys = df[[\"level\", \"name\", \"event_name\", \"room_fqid\", \"fqid\", \"text\"]].values.tolist()\n        keys = [str(values[0]).zfill(2) + \"_\" + \"_\".join([\"None\" if type(v) != str else v for v in values[1:]]) for values in keys]\n        count_keys = Counter(keys)\n        valid_keys = [key for key, value in count_keys.items() if value >= threshold]\n        self.map_key = {key: i for i, key in enumerate(valid_keys)}\n        self.valid_keys = set(valid_keys)\n        \n        # text sequence of important events (notification_click)\n        df_tmp = df.query(\"event_name == 'notification_click'\")[[\"session_id\", \"text\"]].fillna(\"\")\n        df_tmp[\"text_prev\"] = df_tmp.groupby(\"session_id\")[\"text\"].shift()\n        df_tmp = df_tmp.dropna()\n        df_agg = df_tmp.groupby([\"text_prev\", \"text\"], as_index=False).size().sort_values(\"size\", ascending=False)\n        self.list_text_seq = [(text1, text2) for text1, text2 in zip(df_agg[\"text_prev\"], df_agg[\"text\"])]\n        self.map_text_seq = {value: i for i, value in enumerate(self.list_text_seq)}\n        \n        self.feature_names += [f\"count_{key}\" for key in self.map_key.keys()]\n        self.feature_names += [f\"bdiff_{key}\" for key in self.map_key.keys()]\n        self.feature_names += [f\"fdiff_{key}\" for key in self.map_key.keys()]\n        self.feature_names += [f\"time_between_{text1}_and_{text2}\" for text1, text2 in self.list_text_seq]\n        self.feature_names += [\"last_time\", \"diff_level_group_0to1\", \"diff_level_group_1to2\"]\n    \n    def make_feature(self, features, df_session, session_id, level_group):\n        keys = df_session[[\"level\", \"name\", \"event_name\", \"room_fqid\", \"fqid\", \"text\"]].values.tolist()\n        keys = [str(values[0]).zfill(2) + \"_\" + \"_\".join([\"None\" if type(v) != str else v for v in values[1:]]) for values in keys]\n        \n        values_time = df_session[\"elapsed_time\"].tolist()\n        values_event = df_session[\"event_name\"].tolist()\n        values_text = df_session[\"text\"].tolist()\n        \n        # sort by index (dealing with api issue)\n        argsort = np.argsort(df_session[\"index\"].values).tolist()\n        keys = list(map(keys.__getitem__, argsort))\n        values_time = list(map(values_time.__getitem__, argsort))\n        values_event = list(map(values_event.__getitem__, argsort))\n        values_text = list(map(values_text.__getitem__, argsort))\n        \n        # time diff between level_group\n        if level_group == 1:\n            features[-2] = values_time[0] - features[-3]\n        elif level_group == 2:\n            features[-1] = values_time[0] - features[-3]\n        \n        # last time\n        features[-3] = values_time[-1]\n\n        text_prev = \"\"\n        time_prev = 0\n        for i in range(len(df_session)):\n            if keys[i] in self.valid_keys:\n                # count\n                feature_idx = self.map_key[keys[i]]\n                if level_group <= 1:\n                    features[feature_idx] += 1\n                # bdiff\n                feature_idx += len(self.map_key)\n                if level_group <= 1 and i > 0:\n                    features[feature_idx] += values_time[i] - values_time[i-1]\n                # fdiff\n                feature_idx += len(self.map_key)\n                if i < len(df_session) - 1:\n                    features[feature_idx] += values_time[i+1] - values_time[i]\n            # time between important events\n            if values_event[i] == \"notification_click\":\n                if (text_prev, values_text[i]) in self.map_text_seq:\n                    feature_idx = len(self.map_key)*3 + self.map_text_seq[(text_prev, values_text[i])]\n                    features[feature_idx] += values_time[i] - time_prev\n                text_prev = values_text[i]\n                time_prev = values_time[i]\n\n        return features","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:43.457500Z","iopub.execute_input":"2023-12-14T02:43:43.459402Z","iopub.status.idle":"2023-12-14T02:43:43.477048Z","shell.execute_reply.started":"2023-12-14T02:43:43.459373Z","shell.execute_reply":"2023-12-14T02:43:43.476336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load initial feature maker\nwith open(MODEL_DIR + \"feature_maker.pickle\", \"rb\") as f:\n    fm = pickle.load(f)\ndict_feature_index = {f: i for i, f in enumerate(fm.feature_names)}","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:43.480874Z","iopub.execute_input":"2023-12-14T02:43:43.482916Z","iopub.status.idle":"2023-12-14T02:43:43.505000Z","shell.execute_reply.started":"2023-12-14T02:43:43.482888Z","shell.execute_reply":"2023-12-14T02:43:43.504322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = []\nidx_features = [[], [], []]\nvalid_keys = set()\nfor level_group in range(3):\n    # Load each model\n    with open(MODEL_DIR + f\"model{level_group}.pickle\", \"rb\") as f:\n        model = pickle.load(f)\n        \n    # Add each model to list of models\n    models.append(model)\n    \n    # Get number of model's features\n    n_features = len(model.feature_name())\n    \n    # Get the important features\n    df_importance = pd.read_csv(MODEL_DIR + f\"importance{level_group}.csv\")\n    df_importance = df_importance.groupby(\"feature\")[\"importance\"].mean().sort_values(ascending=False)\n    features = df_importance.index[1:n_features-1].tolist()\n    idx_features[level_group] = [dict_feature_index[f] for f in features]\n    \n    # Get valid keys\n    valid_keys |= set([f[6:] for f in features if f[:5] in [\"count\", \"bdiff\", \"fdiff\"]])","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:43.508623Z","iopub.execute_input":"2023-12-14T02:43:43.510413Z","iopub.status.idle":"2023-12-14T02:43:46.410509Z","shell.execute_reply.started":"2023-12-14T02:43:43.510385Z","shell.execute_reply":"2023-12-14T02:43:46.409802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set valid keys for model\nfm.valid_keys = valid_keys","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:46.411788Z","iopub.execute_input":"2023-12-14T02:43:46.412236Z","iopub.status.idle":"2023-12-14T02:43:46.416442Z","shell.execute_reply.started":"2023-12-14T02:43:46.412210Z","shell.execute_reply":"2023-12-14T02:43:46.415153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize arrays storing number of questions of each level group and corresponding questions\nn_questions = [3, 10, 5]\nquestions = [\n    np.array([1, 2, 3], dtype=np.float32),\n    np.array([4, 5, 6, 7, 8, 9, 10, 11, 12, 13], dtype=np.float32),\n    np.array([14, 15, 16, 17, 18], dtype=np.float32)\n]","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:46.417991Z","iopub.execute_input":"2023-12-14T02:43:46.418278Z","iopub.status.idle":"2023-12-14T02:43:46.429312Z","shell.execute_reply.started":"2023-12-14T02:43:46.418252Z","shell.execute_reply":"2023-12-14T02:43:46.428330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set threshold to 0.625\nthreshold = 0.625","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:46.432143Z","iopub.execute_input":"2023-12-14T02:43:46.432432Z","iopub.status.idle":"2023-12-14T02:43:46.441833Z","shell.execute_reply.started":"2023-12-14T02:43:46.432407Z","shell.execute_reply":"2023-12-14T02:43:46.440523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_features = dict()\nfor df_session, df_pred in iter_test:\n    # Get session id of the current test session dataframe\n    session_id = df_session.iloc[0, 0]\n    \n    # If length of df_pred is 3, 10 or 5 then assign level group to 0, 1 or 2, respectively\n    if len(df_pred) == 3:\n        level_group = 0\n    elif len(df_pred) == 10:\n        level_group = 1\n    elif len(df_pred) == 5:\n        level_group = 2\n    \n    if level_group == 0:\n        # Initialize value of `dict_features` at key `session_id`\n        dict_features[session_id] = [0] * len(fm.feature_names)\n    \n    # Update value of `dict_features` at key `session_id` \n    dict_features[session_id] = fm.make_feature(dict_features[session_id], df_session, session_id, level_group)\n    \n    # Get test set and use model to predict\n    X_test = list(map(dict_features[session_id].__getitem__, idx_features[level_group]))\n    X_test = np.array([0] + [value if value != 0 else np.nan for value in X_test] + [22], dtype=np.float32)  # append \"q\" and \"level_max\"\n    X_test = np.tile(X_test, (n_questions[level_group], 1))\n    X_test[:, 0] = questions[level_group]\n    preds = (models[level_group].predict(X_test) > threshold).astype(int)\n    \n    # Overwrite the session_id values (dealing with api issue)\n    df_pred[\"session_id\"] = [f\"{session_id}_q{int(q)}\" for q in questions[level_group]]\n    df_pred[\"correct\"] = preds\n    env.predict(df_pred)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:46.444162Z","iopub.execute_input":"2023-12-14T02:43:46.444525Z","iopub.status.idle":"2023-12-14T02:43:46.548616Z","shell.execute_reply.started":"2023-12-14T02:43:46.444492Z","shell.execute_reply":"2023-12-14T02:43:46.547525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print results in `submission.csv`\n! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:46.549931Z","iopub.execute_input":"2023-12-14T02:43:46.550243Z","iopub.status.idle":"2023-12-14T02:43:46.874292Z","shell.execute_reply.started":"2023-12-14T02:43:46.550215Z","shell.execute_reply":"2023-12-14T02:43:46.872664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count number of correct and in correct answer being predicted\ndf_pred = pd.read_csv(\"submission.csv\")\ndf_pred[\"correct\"].value_counts()\n\nunique, count = np.unique(df_pred['correct'], return_counts=True)\nunique = unique.astype('int')\n\nfig, ax = plt.subplots(figsize=(5,5))\nbars = ax.bar([0, 0.25], count, color=['b','c'], width=0.1)\nax.set_xticks([0, 0.25], unique)\nax.bar_label(bars, fmt='{:.0f}')\n\nax.set_xlabel('Correctness')\nax.set_ylabel('Count')\nax.set_title('Number of correct and incorrect answers');","metadata":{"execution":{"iopub.status.busy":"2023-12-14T02:43:46.878509Z","iopub.execute_input":"2023-12-14T02:43:46.878839Z","iopub.status.idle":"2023-12-14T02:43:47.254917Z","shell.execute_reply.started":"2023-12-14T02:43:46.878813Z","shell.execute_reply":"2023-12-14T02:43:47.253812Z"},"trusted":true},"execution_count":null,"outputs":[]}]}