{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import riiideducation\nimport pandas as pd\n\npath_train_csv = '../input/riiid-test-answer-prediction/train.csv'\npath_questions_csv = '../input/riiid-test-answer-prediction/questions.csv'\n\nd_types = {'row_id': 'int64', #  0\n           'timestamp': 'int64',#  1\n           'user_id': 'int32', #  2\n           'content_id': 'int16',#  3\n           'content_type_id': 'int8',#  4\n           'task_container_id': 'int16',#  5\n           'user_answer': 'int8', #  6\n           'answered_correctly': 'int8',#  7\n           'prior_question_elapsed_time': 'float32',#  8 \n           'prior_question_had_explanation': 'boolean',#  9\n         }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# read csv\ntrain_df = pd.read_csv(path_train_csv,\n                       dtype=d_types,\n                       usecols=[2, 3, 4, 6]\n                      )\ntarget_df = train_df.query('content_type_id == 0')[['content_id', 'user_id', 'user_answer']]\ndel train_df\n# obtain first trials\nuser_answer_df = target_df.groupby(['content_id', 'user_id']).first().user_answer\ndel target_df\n# grouped by questions and obtain user reactions on their first trials\nanswer_hist_df = user_answer_df.groupby('content_id').value_counts()\ndel user_answer_df\nanswer_hist_df_ = answer_hist_df.unstack()\ndel answer_hist_df\nanswer_hist_df = answer_hist_df_.fillna(0).astype('int32')\nanswer_hist_df['total'] = answer_hist_df.sum(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# join question data and culc correct rates\nq_data_df = pd.read_csv(path_questions_csv)\nq_df = pd.merge(q_data_df, answer_hist_df, left_on='question_id', right_on='content_id')\nq_df['answered_correctly'] = q_df.apply(lambda x : x[x['correct_answer']] / x['total'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# put correct rate values as predicted \nenv = riiideducation.make_env()\niter_test = env.iter_test()\nfor (test_df, sample_prediction_df) in iter_test:\n    t_test_df = pd.merge(test_df.query('content_type_id == 0'), q_df, right_on='question_id', left_on='content_id')[['row_id', 'answered_correctly']]\n    env.predict(t_test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}