{"cells":[{"metadata":{},"cell_type":"markdown","source":"Heatmap of the first thirty questions to all users (based on 10M records).\n\nThe standard questions and the bundles are easy to identify.  The patterns seem to continue (in increments of 30) but in much smaller quantities.  If you play with this, pay attention to changes in the scale, it changes dramatically.\n\nI wanted to get this further along but didn't want to wait any longer.\n\nHappy New Year!"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib.ticker import FuncFormatter\nimport pandas as pd\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', nrows=10_000_000)\ndf = df.loc[df.content_type_id == 0].reset_index(drop=True) # drop lectures\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\ndf['counter'] = 1\ndf['user_ques_num'] = df.groupby('user_id').cumcount()+1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def questions_heatmap(df, start_ques=0, num_of_ques=30, start_content_id_rank=0, num_of_content_ids=70, figsize=(10,10), show_avg=False, scb=.99):\n\n    df_stats = df.groupby(['content_id'])[['counter']].sum().sort_values(['counter'], ascending=False).reset_index().reset_index().rename(columns={'index':'content_id_rank', 'counter':'content_id_attempts'})\n    df2 = df.merge(df_stats[['content_id', 'content_id_rank', 'content_id_attempts']], left_on='content_id', right_on='content_id', how='left')\n\n    df2 = df2[((df2['user_ques_num']>= start_ques) & (df2['user_ques_num']<= start_ques + num_of_ques))] # limit user questions\n    df2 = df2[((df2['content_id_rank']>=start_content_id_rank) & (df2['content_id_rank']<=(start_content_id_rank+num_of_content_ids)))].reset_index()\n\n    df2 = df2.groupby(['content_id', 'user_ques_num', 'content_id_rank', 'content_id_attempts'])[['counter', 'answered_correctly']].sum() # .sort_values(['counter'], ascending=False)\n    df2['cont_set_ques_pct'] = df2['answered_correctly'] / df2['counter']\n    df2.drop(['answered_correctly'], axis=1, inplace=True)\n    df2 = pd.DataFrame(df2.unstack(level=1)).sort_values(['content_id_rank'], ascending=[True]).fillna(0)\n    \n    new_y_axis_tms = zip(list(df2.index.get_level_values(0)), list(df2.index.get_level_values(1)), list(df2.index.get_level_values(2)))\n    \n    cols = [col for col in df2.columns if col[0] == 'cont_set_ques_pct']\n    if show_avg:\n        x = df2[cols].to_numpy()\n        y = np.array([f\"{w:.2f}\".lstrip('0') for w in x.reshape(x.size)])\n        y = y.reshape(x.shape)\n    else:\n        y = False\n    df2.drop(cols, axis=1, inplace=True)\n    df2.columns = list(range(cols[0][1], df2.shape[1] + cols[0][1]))\n\n    fig = plt.figure(figsize=figsize)\n    ax1 = plt.subplot2grid((20, 20), (0, 0), colspan=20, rowspan=20)\n    comma_fmt = FuncFormatter(lambda x, p: format(int(x), ','))\n    sns.heatmap(df2, \n                ax=ax1, \n                cmap='gist_heat_r', \n                square=True,\n                annot=y, \n                cbar_kws={\"shrink\": scb, 'format': comma_fmt},\n                annot_kws={\"fontsize\":8}, \n                fmt='');\n    plt.xlabel(\"question number\", fontsize=15)\n    plt.ylabel(\"( Content id  /  Times Asked Rank  /  # Times Asked )\", fontsize=15)\n    \n#     ax1.set_yticklabels(new_y_axis_tms)\n    ax1.set_title('Riiid Question Heatmap\\n', fontsize=20)\n    plt.yticks(rotation=0)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_heatmap(df, start_ques=0, num_of_ques=30, start_content_id_rank=0, num_of_content_ids=44, figsize = (10,15), show_avg=False, scb=.8 )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Below zooms out showing the first two hunderd questions and content id's using the same scale as above."},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_heatmap(df, start_ques=0, num_of_ques=200, start_content_id_rank=0, num_of_content_ids=200, figsize=(20,25), show_avg=False, scb=.65)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}