{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport psutil\nfrom time import time\nfrom contextlib import contextmanager\nfrom tqdm.notebook import tqdm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"dtypes = {'timestamp': 'int64', \n          'user_id': 'int32' ,\n          'content_id': 'int16',\n          'content_type_id': 'int8',\n          'answered_correctly':'int8'}\ntrain_cols = ['timestamp', \n              'user_id', \n              'content_id', \n              'content_type_id', \n              'answered_correctly']\n\ntrain_df = pd.read_pickle('../input/cv-strategy/cv4_train.pickle')\ntrain_df = train_df[train_cols]\ntrain_df = train_df.astype(dtypes)\n\ntrain_df = train_df[train_df.content_type_id == False]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ngroup = train_df[['user_id','content_id']].groupby('user_id').apply(lambda r: (r['content_id'].values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ngroup = train_df.groupby('user_id')['content_id'].apply(lambda r: (r.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_pickle('../input/cv-strategy/cv2_valid.pickle')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class Iter_Valid(object):\n    '''\n    https://www.kaggle.com/its7171/time-series-api-iter-test-emulator\n    '''\n    def __init__(self, df, max_user=1000):\n        df = df.reset_index(drop=True)\n        self.df = df\n        self.user_answer = df['user_answer'].astype(str).values\n        self.answered_correctly = df['answered_correctly'].astype(str).values\n        df['prior_group_responses'] = \"[]\"\n        df['prior_group_answers_correct'] = \"[]\"\n        self.sample_df = df[df['content_type_id'] == 0][['row_id']]\n        self.sample_df['answered_correctly'] = 0\n        self.len = len(df)\n        self.user_id = df.user_id.values\n        self.task_container_id = df.task_container_id.values\n        self.content_type_id = df.content_type_id.values\n        self.max_user = max_user\n        self.current = 0\n        self.pre_user_answer_list = []\n        self.pre_answered_correctly_list = []\n\n    def __iter__(self):\n        return self\n    \n    def fix_df(self, user_answer_list, answered_correctly_list, pre_start):\n        df= self.df[pre_start:self.current].copy()\n        sample_df = self.sample_df[pre_start:self.current].copy()\n        df.loc[pre_start,'prior_group_responses'] = '[' + \",\".join(self.pre_user_answer_list) + ']'\n        df.loc[pre_start,'prior_group_answers_correct'] = '[' + \",\".join(self.pre_answered_correctly_list) + ']'\n        self.pre_user_answer_list = user_answer_list\n        self.pre_answered_correctly_list = answered_correctly_list\n        return df, sample_df\n\n    def __next__(self):\n        added_user = set()\n        pre_start = self.current\n        pre_added_user = -1\n        pre_task_container_id = -1\n\n        user_answer_list = []\n        answered_correctly_list = []\n        while self.current < self.len:\n            crr_user_id = self.user_id[self.current]\n            crr_task_container_id = self.task_container_id[self.current]\n            crr_content_type_id = self.content_type_id[self.current]\n            if crr_content_type_id == 1:\n                # no more than one task_container_id of \"questions\" from any single user\n                # so we only care for content_type_id == 0 to break loop\n                user_answer_list.append(self.user_answer[self.current])\n                answered_correctly_list.append(self.answered_correctly[self.current])\n                self.current += 1\n                continue\n            if crr_user_id in added_user and ((crr_user_id != pre_added_user) or (crr_task_container_id != pre_task_container_id)):\n                # known user(not prev user or differnt task container)\n                return self.fix_df(user_answer_list, answered_correctly_list, pre_start)\n            if len(added_user) == self.max_user:\n                if  crr_user_id == pre_added_user and crr_task_container_id == pre_task_container_id:\n                    user_answer_list.append(self.user_answer[self.current])\n                    answered_correctly_list.append(self.answered_correctly[self.current])\n                    self.current += 1\n                    continue\n                else:\n                    return self.fix_df(user_answer_list, answered_correctly_list, pre_start)\n            added_user.add(crr_user_id)\n            pre_added_user = crr_user_id\n            pre_task_container_id = crr_task_container_id\n            user_answer_list.append(self.user_answer[self.current])\n            answered_correctly_list.append(self.answered_correctly[self.current])\n            self.current += 1\n        if pre_start < self.current:\n            return self.fix_df(user_answer_list, answered_correctly_list, pre_start)\n        else:\n            raise StopIteration()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"iter_test = Iter_Valid(test_df,max_user=1000)\npredicted = []\ndef set_predict(df):\n    predicted.append(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\npbar = tqdm(total=len(test_df))\nprevious_test_df = None\nfor (current_test, current_prediction_df) in iter_test:\n    if previous_test_df is not None:\n        answers = eval(current_test[\"prior_group_answers_correct\"].iloc[0])\n        responses = eval(current_test[\"prior_group_responses\"].iloc[0])\n        previous_test_df['answered_correctly'] = answers\n        previous_test_df['user_answer'] = responses\n        prev_group = previous_test_df[['user_id', 'content_id']]\\\n        .groupby('user_id').apply(lambda r: (\n        r['content_id'].values))\n        \n        \n    previous_test_df = current_test.copy()\n    current_test = current_test[current_test.content_type_id == 0]\n    # your prediction code here\n    current_test['answered_correctly'] = 0.5\n    set_predict(current_test.loc[:,['row_id', 'answered_correctly']])\n    pbar.update(len(current_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"iter_test = Iter_Valid(test_df,max_user=1000)\npredicted = []\ndef set_predict(df):\n    predicted.append(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\npbar = tqdm(total=len(test_df))\nprevious_test_df = None\nfor (current_test, current_prediction_df) in iter_test:\n    if previous_test_df is not None:\n        answers = eval(current_test[\"prior_group_answers_correct\"].iloc[0])\n        responses = eval(current_test[\"prior_group_responses\"].iloc[0])\n        previous_test_df['answered_correctly'] = answers\n        previous_test_df['user_answer'] = responses\n        prev_group = previous_test_df[['user_id', 'content_id']]\\\n        .groupby('user_id')['content_id'].apply(lambda r: (\n        r.values))\n        \n        \n    previous_test_df = current_test.copy()\n    current_test = current_test[current_test.content_type_id == 0]\n    # your prediction code here\n    current_test['answered_correctly'] = 0.5\n    set_predict(current_test.loc[:,['row_id', 'answered_correctly']])\n    pbar.update(len(current_test))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}