{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport gc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set(style='darkgrid')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# loading the data\ntrain = pd.read_feather('../input/riiid-train-data-multiple-formats/riiid_train.feather', \n                        columns=['user_id', 'timestamp', 'user_answer', 'answered_correctly'])\nquestions = pd.read_csv('../input/riiid-test-answer-prediction/questions.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Time elapsed since the user started using the app"},{"metadata":{"trusted":true},"cell_type":"code","source":"# aggregation(max value of timestamp per user) \nagg_df = train.groupby('user_id')['timestamp'].max().reset_index()\nagg_df = agg_df.sort_values('timestamp', ascending=False) \nagg_df['day'] = agg_df['timestamp'] / (1000*60*60*24)\nagg_df['day'].describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Most are used for a few days or months.  \nbut the longest users have been using it for more than 1000days!"},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,4))\nplt.hist(agg_df['day'])\nplt.xlabel('elapsed days')\nplt.ylabel('number of users')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"They're either continuous learners or, if they're not, they're app developers."},{"metadata":{},"cell_type":"markdown","source":"# correct answer to all questions"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train.answered_correctly>=0]\n\n# Calculate the percentage of correct answers per user\nagg_df = train.groupby('user_id')['answered_correctly'].agg(['count', 'sum']).reset_index()\nagg_df.columns = ['user_id', 'total_answer_cnt', 'correct_cnt']\nagg_df['correct_rate'] = agg_df['correct_cnt'] / agg_df['total_answer_cnt']\nagg_df.sort_values(['correct_rate', 'total_answer_cnt'], ascending=False).head(20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"user_id:48102728 answered all 41 questions correctly.  \nbut the numbers aren't surprising. (The real TOEIC test has 200 questions.)"},{"metadata":{},"cell_type":"markdown","source":"# users who keep choosing the same answer"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Calculate the percentage of each answer number selected by each user.\nagg_df = train.groupby(['user_id', 'user_answer'])['answered_correctly'].count().reset_index()\nagg_df.columns = ['user_id', 'user_answer', 'answer_cnt']\nagg_df['total_answer_cnt'] = agg_df.groupby('user_id').transform('sum')['answer_cnt']\nagg_df['selection_rate'] = agg_df['answer_cnt'] / agg_df['total_answer_cnt']\nagg_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_df.sort_values(['selection_rate', 'total_answer_cnt'], ascending=False).head(20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are quite a few users who always pick the same number.  \nAre they abandoning answers?"},{"metadata":{},"cell_type":"markdown","source":"# by the way,  \nI noticed a things in the investigation above.  \nMany users with total_answer_cnt of 30.  \nLet us just check one last thing."},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_df = train.groupby('user_id')['user_answer'].count().reset_index()\nagg_df.columns = ['user_id', 'total_answer_cnt']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vc = agg_df['total_answer_cnt'].value_counts()\nvc.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,4))\nplt.bar(vc.index, vc)\nplt.xlim(0,100)\nplt.xlabel('total answer count')\nplt.ylabel('number of users')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"(total_answer_cnt=30) are remarkably high.  \nAnd then there are round number like 40 and 50.  \n\nThere may be a total of 30 questions to test user skills when to start using the app.  \nBasically are there 10 questions per a attempt?"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}