{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Repetitions in Riiid\n\nMany users took tests for multiple times.\n\nBelow you can see effect of repetitions for a single bundle."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport re\nfrom collections import defaultdict\nimport pickle\nimport gc\n\nimport numpy as np\nimport pandas as pd\nimport psutil\nfrom pylab import plt, plot, hist, legend, array, arange, zeros, ones, sqrt, where, cm\nimport seaborn as sns\n\n_ = np.seterr(divide='ignore', invalid='ignore')\n\nfrom tqdm import tqdm\n# tqdm.pandas()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"PATH = '../input/riiid-test-answer-prediction/'\ncol_dtype = {'row_id': 'int64', \n             'timestamp': 'int64', \n             'user_id': 'int32', \n             'content_id': 'int16', \n             'content_type_id': 'int8',\n             'task_container_id': 'int16', \n             'user_answer': 'int8', \n             'answered_correctly': 'int8', \n             'prior_question_elapsed_time': 'float32', \n             'prior_question_had_explanation': 'boolean',\n             }\nSECPERD = 24*60*60\n\nTARGET = 'answered_correctly'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Q = pd.read_csv(PATH+'questions.csv', engine='c')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"D = pd.read_csv(PATH+'train.csv', engine='c'#, nrows=700_000\n               )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's select questions in bundles for stereotype check"},{"metadata":{"trusted":true},"cell_type":"code","source":"QN = Q.bundle_id.value_counts()\n\nQN[QN==5].head(20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Typical trajectory for 1 bundle"},{"metadata":{"trusted":true},"cell_type":"code","source":"qid = 7780\nbid = Q.loc[Q.question_id==qid,'bundle_id'].iloc[0]\nz = Q.loc[Q.question_id==qid,'part'].iloc[0]\n\nprint(D[D.content_id==qid].content_type_id.value_counts(dropna=False))\naa = Q.loc[Q.bundle_id==bid, 'correct_answer']\nnb = len(aa)\n\n\nqii = qid + arange(nb)\nX = D[(D.content_type_id==0) & (D.content_id.isin(qii))]\n\nuagg = X.iloc[-1_000:].groupby(['user_id','task_container_id'])\n\n# xb = X.iloc[-nb:]\n# xb\n\nfor (uid,tcid), xb in iter(uagg):\n    if len(xb)!= nb:\n        logger.warning(f'? {len(xb)} of {nb} records (u={uid},tc={tcid}) ')\n#         continue\n    i = uid\n    co = plt.cm.jet(i%256, alpha=.05)\n    xb.sort_values('content_id', inplace=True)\n    bbcorr = (xb.answered_correctly.values > 0)\n    ii = arange(len(bbcorr))+1\n    plot(ii[~bbcorr], xb.user_answer[~bbcorr], color=co, marker='x', lw=0)\n    plot(ii[bbcorr], xb.user_answer[bbcorr], color=co, marker='o', lw=0)\n    plot(ii, xb.user_answer, color=co, marker=None, lw=5)\n    \nplot(ii, aa, marker='o', ms=30, mec='k', mfc='None', lw=0)    \nplt.xticks(ii, qii)\nplt.box(False)\nplt.title(f'{len(uagg)} answers on bundle {qid}, part {z}');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"not all users have data for complete bundle:  they skipped some questions or the data are fragmented\n\n(for the whole dataset it is true)"},{"metadata":{"trusted":true},"cell_type":"code","source":"uagg = X.sort_values('content_id').groupby(['user_id','task_container_id'], as_index=True)\nutn = uagg.row_id.count()\nutn","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Repeats"},{"metadata":{"trusted":true},"cell_type":"code","source":"u_ntc = utn.reset_index().groupby('user_id')['task_container_id'].count()\nu_ntc[u_ntc>1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"u_ntc.max()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Up to 8 repetitions!"},{"metadata":{},"cell_type":"markdown","source":"It is interesting, whether tcid is related with order of training for every user?"},{"metadata":{"trusted":true},"cell_type":"code","source":"tcpo = utn.reset_index()['task_container_id'].value_counts().sort_index()\n\ntcpo.rolling(50, 1).sum().plot();","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Can timestamp differ for a single bundle?"},{"metadata":{"trusted":true},"cell_type":"code","source":"mm_utc = uagg['timestamp'].agg(['min','max'])\n\nnp.all(mm_utc['max'] == mm_utc['min'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"No!  Bundle of questions is momental in time!\n\nSo we dont know the order ... may be they are randomised, but we assume the order as question_id grows."},{"metadata":{},"cell_type":"markdown","source":"Balls by user and task_container ..."},{"metadata":{"trusted":true},"cell_type":"code","source":"b_utc = uagg['answered_correctly'].sum()\nb_utc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"nrep=6\nbagg = b_utc[u_ntc[u_ntc>2].index].droplevel(1).reset_index().groupby('user_id')['answered_correctly']\n\nbrep = pd.concat([bagg.nth(i) for i in range(nrep)], 1)\nbrep.columns=range(nrep)\nbrep","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.errorbar(arange(nrep), brep.mean(), brep.std());\nplt.xlabel('Attempt')\nplt.ylabel(f'Q of {nb}')\nplt.title(f'Progress for bundle {bid}');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"for poor starters only ..."},{"metadata":{"trusted":true},"cell_type":"code","source":"brep_ = brep[brep[0]<3]\nplt.errorbar(arange(nrep), brep_.mean(), brep_.std());\nplt.xlabel('Attempt')\nplt.ylabel(f'Q of {nb}')\nplt.title(f'Progress for bundle {bid}');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Always dummy ?\n\nWhether some users click on the same answer option?"},{"metadata":{"trusted":true},"cell_type":"code","source":"uaa = uagg.apply(lambda x: None if len(x) < nb else pd.Series(x['user_answer'].values, index=qii, name=x.index[0]))\nuaa","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n = uaa.shape[0]\n\nncorr = len(uaa[(uaa == aa.values).all(1)])\n\nncorr/n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Only 22% of users completed the bundle w/o errors!"},{"metadata":{"trusted":true},"cell_type":"code","source":"((uaa == aa).sum() / n).plot(kind='barh'); \nplt.xlim(0,1);\nplt.title('% of correct anser per question');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Random guess answers (probability=0.25) are included!\n\nBut also partitial knowledge...\n\nAbsolute dummies are those who click the same variant."},{"metadata":{"trusted":true},"cell_type":"code","source":"traj = np.ones(uaa.shape[1]) * 0\n\nuaa[(uaa == traj).all(1)]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There were such dummy users!\n\nLet's count percentage..."},{"metadata":{"trusted":true},"cell_type":"code","source":"{i: '{:.2%}'.format(sum((uaa == (np.ones(uaa.shape[1]) * i)).all(1)) / n ) for i in range(4) }","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The second variant is more even popular!\n\n... or they may randomised in test presentation software?\n\nKoreans who took TOEIC, write in comments!"},{"metadata":{},"cell_type":"markdown","source":"## Conclusion\n\nYou can trace repetition and give higher predictions for repeated measures or craft an additional feature for machine learning!\n\nYou can mark those users or sessions as dummy style, and predict 0.25 - the probability for dummy choice of 4.\n\nBut you must repeat this analysis for every question!"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}