{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Version history\n1. Initial version\n2. Added `_state` & `State` thanks to @qhapaq49's comment\n3. labels & train got swapped in the official library. Swapping it here as well.\n   (FYI: [more data available here - Click on Jowilder on the left side](https://fielddaylab.wisc.edu/opengamedata/)\n4. Fix a small error in the last cell\n5. Fix an error that might appear if you choose to use the training data as input data (in case you want a bit more than the 'test.csv' which was provided).\n\n\n\n\n### Tips for re-running your notebook without a restart:\n```python\nfrom jo_wilder import make_env\nmake_env.__called__ = False  # 1: be able to call make_env() again\n\ncompetition = \n\nfor train, labels in make_env().iter_test():\n    # run model ... do something...\n    ...\n    # predict the results\n    competition.predict(labels)\n```","metadata":{}},{"cell_type":"code","source":"import sys\n\nimport pandas as pd\nimport numpy as np\nfrom enum import Enum\n\nfrom itertools import repeat\n\nif sys.platform == 'win32':  # use Mac OSX => 'darwin', Linux => 'linux'\n    _input_dir = 'data'\n    _output_dir = '.'\nelse:\n    _input_dir = '/kaggle/input/predict-student-performance-from-game-play'\n    _output_dir = '/kaggle/working'\n\n    \nclass State(Enum):\n    INIT = 1\n    AWAITING_PREDICT = 2\n    MADE_PREDICTION = 3\n    DONE = 4\n\n    \ndef make_env() -> \"Competition\":\n    if make_env.__called__:\n        raise Exception(\"You can only call `make_env()` once.\")\n\n    make_env.__called__ = True\n    return Competition(pd.read_csv(f'{_input_dir}/test.csv'))\n\nmake_env.__called__ = False\n\n\nclass Competition:\n    _state: State = State.INIT\n\n    groups = {\n        '0-4': list(range(1, 4)),\n        '5-12': list(range(4, 14)),\n        '13-22': list(range(14, 19)),\n    }\n\n    def __init__(self, df):\n        df['level_group'] = pd.Categorical(df.level_group, list(self.groups))\n        \n        self.df = df\n        \n        df_groupby = self.df.sort_values(['level_group', 'session_id']).groupby(['level_group', 'session_id'])\n        self.df_iter = df_groupby.__iter__()\n\n        self.predictions = None\n\n    def __iter__(self):\n        return self\n\n    def iter_test(self):\n        return self\n\n    def __next__(self):\n        assert self._state in [State.INIT, State.MADE_PREDICTION], \"You must call `predict()` before you get the next batch of data.\"\n\n        try:\n            (level_group, session_id), df = next(self.df_iter)\n        except StopIteration:\n            self._state = State.DONE\n            self.predictions.to_csv('local_submission.csv', index=False)\n            raise\n\n        pred_df = pd.DataFrame({\n            'session_id': [f'{session_id}_q{q}' for q in self.groups[level_group]],\n            'correct': [0 for _ in self.groups[level_group]],\n        })\n\n        self._state = State.AWAITING_PREDICT\n        if 'session_level' in df.columns:\n            df = df.drop(columns=['session_level'])\n\n        return df.reset_index(drop=True), pred_df\n\n    def predict(self, pred_df):\n        assert self._state == State.AWAITING_PREDICT, \"You must get the next batch before making a new prediction.\"\n        assert pred_df.columns.to_list() == ['session_id', 'correct'], \"Prediction dataframe have invalid columns.\"\n\n        if self.predictions is not None:\n            self.predictions = pd.concat([self.predictions, pred_df])\n        else:\n            self.predictions = pred_df.copy()\n\n        self._state = State.MADE_PREDICTION","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-03-13T04:59:13.272996Z","iopub.execute_input":"2023-03-13T04:59:13.273391Z","iopub.status.idle":"2023-03-13T04:59:13.289817Z","shell.execute_reply.started":"2023-03-13T04:59:13.273355Z","shell.execute_reply":"2023-03-13T04:59:13.288603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from jo_wilder import make_env  # <-- toggle this to switch to the real submission algorithm\n\nmake_env.__called__ = False\n\ncompetition = make_env()\nfor train, labels in competition.iter_test():\n    # run model ... do something...\n    ...\n    # predict the results\n    competition.predict(labels)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T04:59:13.291515Z","iopub.execute_input":"2023-03-13T04:59:13.291902Z","iopub.status.idle":"2023-03-13T04:59:13.343568Z","shell.execute_reply.started":"2023-03-13T04:59:13.291871Z","shell.execute_reply":"2023-03-13T04:59:13.342639Z"},"trusted":true},"execution_count":null,"outputs":[]}]}