{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from jo_wilder import make_env\nimport pickle\nimport json\nfrom sklearn.preprocessing import LabelEncoder\nfrom pprint import pprint\nimport pandas as pd\nfrom xgboost import XGBClassifier\nimport numpy as np\n","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:04.586392Z","iopub.execute_input":"2023-03-19T20:41:04.587205Z","iopub.status.idle":"2023-03-19T20:41:04.593574Z","shell.execute_reply.started":"2023-03-19T20:41:04.587154Z","shell.execute_reply":"2023-03-19T20:41:04.592150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"!pip install memory_profiler\n\n%load_ext memory_profiler","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:38:25.239389Z","iopub.execute_input":"2023-03-19T20:38:25.239764Z"}}},{"cell_type":"code","source":"import pickle\n\nwith open('/kaggle/input/predict-student-performance-from-gameplay/ordinal_encoder.pickle', 'rb') as fp:\n    ordinals = pickle.load(fp)\nordinals.dtype = np.float16\nordinals","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:04.599798Z","iopub.execute_input":"2023-03-19T20:41:04.600185Z","iopub.status.idle":"2023-03-19T20:41:04.613725Z","shell.execute_reply.started":"2023-03-19T20:41:04.600151Z","shell.execute_reply":"2023-03-19T20:41:04.612457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/predict-student-performance-from-gameplay/object_columns.json', 'r') as fp:\n    object_columns = json.load(fp)\nobject_columns","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:04.616214Z","iopub.execute_input":"2023-03-19T20:41:04.617240Z","iopub.status.idle":"2023-03-19T20:41:04.624837Z","shell.execute_reply.started":"2023-03-19T20:41:04.617199Z","shell.execute_reply":"2023-03-19T20:41:04.623808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/predict-student-performance-from-gameplay/best_threshold.json', 'r') as fp:\n    best_threshold = json.load(fp)\nbest_threshold","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:04.626459Z","iopub.execute_input":"2023-03-19T20:41:04.627164Z","iopub.status.idle":"2023-03-19T20:41:04.640740Z","shell.execute_reply.started":"2023-03-19T20:41:04.627126Z","shell.execute_reply":"2023-03-19T20:41:04.639597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nfor question in range(18):\n    question += 1\n    models[question] = XGBClassifier()\n    models[question].load_model(f'/kaggle/input/predict-student-performance-from-gameplay/best_model_{question}.json')","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:04.643306Z","iopub.execute_input":"2023-03-19T20:41:04.644084Z","iopub.status.idle":"2023-03-19T20:41:05.507931Z","shell.execute_reply.started":"2023-03-19T20:41:04.644040Z","shell.execute_reply":"2023-03-19T20:41:05.506659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"import sys\n\nimport pandas as pd\nimport numpy as np\nfrom enum import Enum\n\nfrom itertools import repeat\n\nif sys.platform == 'win32':\n    _input_dir = 'data'\n    _output_dir = '.'\nelse:\n    _input_dir = '/kaggle/input/predict-student-performance-from-game-play'\n    _output_dir = '/kaggle/working'\n\n    \nclass State(Enum):\n    INIT = 1\n    AWAITING_PREDICT = 2\n    MADE_PREDICTION = 3\n    DONE = 4\n\n    \ndef make_env() -> \"Competition\":\n    if make_env.__called__:\n        raise Exception(\"You can only call `make_env()` once.\")\n\n    make_env.__called__ = True\n    return Competition(pd.read_csv(f'{_input_dir}/train.csv'))\n\nmake_env.__called__ = False\n\n\nclass Competition:\n    _state: State = State.INIT\n\n    groups = {\n        '0-4': list(range(1, 4)),\n        '5-12': list(range(4, 14)),\n        '13-22': list(range(14, 19)),\n    }\n\n    def __init__(self, df):\n        df['level_group'] = pd.Categorical(df.level_group, list(self.groups))\n        \n        self.df = df\n        \n        df_groupby = self.df.sort_values(['level_group', 'session_id']).groupby(['level_group', 'session_id'])\n        self.df_iter = df_groupby.__iter__()\n\n        self.predictions = None\n\n    def __iter__(self):\n        return self\n\n    def iter_test(self):\n        return self\n\n    def __next__(self):\n        assert self._state in [State.INIT, State.MADE_PREDICTION], \"You must call `predict()` before you get the next batch of data.\"\n\n        try:\n            (level_group, session_id), df = next(self.df_iter)\n        except StopIteration:\n            self._state = State.DONE\n            self.predictions.to_csv('local_submission.csv', index=False)\n            raise\n\n        pred_df = pd.DataFrame({\n            'session_id': [f'{session_id}_q{q}' for q in self.groups[level_group]],\n            'correct': [0 for _ in self.groups[level_group]],\n        })\n\n        self._state = State.AWAITING_PREDICT\n\n        if 'session_level' in df.columns:\n            df.drop(columns=['session_level'], inplace=True)\n\n        return pred_df, df.reset_index(drop=True)\n\n    def predict(self, pred_df):\n        assert self._state == State.AWAITING_PREDICT, \"You must get the next batch before making a new prediction.\"\n        assert pred_df.columns.to_list() == ['session_id', 'correct'], \"Prediction dataframe have invalid columns.\"\n\n        if self.predictions is not None:\n            self.predictions = pd.concat([self.predictions, pred_df])\n        else:\n            self.predictions = pred_df.copy()\n\n        self._state = State.MADE_PREDICTION","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:20:50.261353Z","iopub.execute_input":"2023-03-19T20:20:50.261697Z","iopub.status.idle":"2023-03-19T20:20:50.284685Z","shell.execute_reply.started":"2023-03-19T20:20:50.261666Z","shell.execute_reply":"2023-03-19T20:20:50.282516Z"}}},{"cell_type":"code","source":"Q4 = ordinals.transform([[0]*6 + ['0-4']])[0,-1]\nQ12 = ordinals.transform([[0]*6 + ['5-12']])[0,-1]\nQ22 = ordinals.transform([[0]*6 + ['13-22']])[0,-1]\n\nimport numpy as np\n\n#Reduce Memory Usage\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(f\">> {mem_usg:.2f} -- Saved {100 * mem_usg / start_mem:0.2f}%\")\n    \n    return df\n\ndef cleanup(df):\n    try:\n        columns_to_drop = ['hover_duration', 'page', 'music', 'fullscreen', 'hq']\n        df.drop(columns_to_drop, axis=1, inplace=True)\n\n        try:\n            df = reduce_memory_usage(df)\n        except:\n            return 0\n\n        df.set_index('session_id', inplace=True)\n\n        try:\n            df[object_columns] = ordinals.transform(df[object_columns])\n            return None\n        except KeyError:\n            return 1\n        except ValueError:\n            return 2\n        except:\n            return 3\n    except:\n        return 4","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:05.509414Z","iopub.execute_input":"2023-03-19T20:41:05.513077Z","iopub.status.idle":"2023-03-19T20:41:05.542007Z","shell.execute_reply.started":"2023-03-19T20:41:05.513030Z","shell.execute_reply":"2023-03-19T20:41:05.540927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"groups = {\n    Q4: (1, 4),\n    Q12: (4, 14),\n    Q22: (14, 19),\n}\ncorrect = {1: 0.7240003395874013,\n 2: 0.9787757874182867,\n 3: 0.9321674165888445,\n 4: 0.7993038458273198,\n 5: 0.5463961287036251,\n 6: 0.7720519568724,\n 7: 0.7292639443076662,\n 8: 0.6143136089651074,\n 9: 0.7354614143815265,\n 10: 0.5003820358264708,\n 11: 0.6441973002801596,\n 12: 0.857458188301214,\n 13: 0.27048136514135324,\n 14: 0.7100772561337975,\n 15: 0.482978181509466,\n 16: 0.7378385261906784,\n 17: 0.6852873758383564,\n 18: 0.9505900331097716}\nimport numpy as np\n\nmake_env.__called__ = False\nenv = make_env()\ntype(env)._state = type(type(env)._state).__dict__['INIT']\n\nresult = None\nfor submission, test in env.iter_test():\n    # print('pre', submission.shape)\n\n    if result is None:\n        result = cleanup(test)\n\n    # print(\"> Result:\", result)\n    \n    if result == 0:  # Score: 0.226\n        pass  # No adjustment\n    elif result == 1:  # Score: 0.414\n        submission['correct'] = 1\n    elif result == 2: # Score: ???\n        submission['correct'] = np.random.randint(0, 2, size=len(submission))\n    elif result == 3:  # Score: 0.587\n        for q in range(1, 19):\n            k = f'_q{q}'\n            if correct[q] >= 0.50:\n                submission.loc[submission.session_id.str.endswith(k), 'correct'] = 1\n            else:\n                submission.loc[submission.session_id.str.endswith(k), 'correct'] = 0\n    elif result == 4:  # Score: 0.547\n        for q in range(1, 19):\n            k = f'_q{q}'\n            if correct[q] >= 0.75:\n                submission.loc[submission.session_id.str.endswith(k), 'correct'] = 1\n            else:\n                submission.loc[submission.session_id.str.endswith(k), 'correct'] = 0\n    elif result == 5:\n        submission['correct'] = 0\n    else:  # Score: 0.648\n        for q in range(1, 19):\n            k = f'_q{q}'\n            if correct[q] >= 0.65:\n                submission.loc[submission.session_id.str.endswith(k), 'correct'] = 1\n            else:\n                submission.loc[submission.session_id.str.endswith(k), 'correct'] = 0\n\n    # print('post', submission.shape)\n\n    env.predict(submission)\n\nprint('Final result:', result)","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:05.544190Z","iopub.execute_input":"2023-03-19T20:41:05.545067Z","iopub.status.idle":"2023-03-19T20:41:05.912491Z","shell.execute_reply.started":"2023-03-19T20:41:05.545021Z","shell.execute_reply":"2023-03-19T20:41:05.911256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"\n    # INFER TEST DATA\n    questions = groups[df.level_group.values[0]]\n\n    for question in range(*questions):\n        try:\n            model = models[question]\n        except KeyError:\n            continue\n\n        guessed = model.predict_proba(df)[:,1].mean()\n        threshold = best_threshold[str(question)]\n        \n        mask = submission.session_id.str.contains(f'q{question}')\n        submission.loc[mask,'correct'] = int(guessed > threshold)\n    \n    env.predict(submission)","metadata":{}},{"cell_type":"code","source":"pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-19T20:41:05.913923Z","iopub.execute_input":"2023-03-19T20:41:05.914372Z","iopub.status.idle":"2023-03-19T20:41:05.930986Z","shell.execute_reply.started":"2023-03-19T20:41:05.914321Z","shell.execute_reply":"2023-03-19T20:41:05.929831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}}]}