{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-06T12:36:09.901018Z","iopub.execute_input":"2023-04-06T12:36:09.901445Z","iopub.status.idle":"2023-04-06T12:36:09.907425Z","shell.execute_reply.started":"2023-04-06T12:36:09.901409Z","shell.execute_reply":"2023-04-06T12:36:09.906163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\" ,\n                       usecols=['session_id', \n                                'index', 'elapsed_time',\n                                'event_name',\n                                'level', 'level_group'],\n                       dtype={'index': np.int16, 'level': np.int8, 'event_name':'category'})\ntrain_df['elapsed_time'] = train_df['elapsed_time']/1000/60\ntrain_df['index2'] = train_df['index'] + 1\n\n\n\ntrain_df = train_df.merge(train_df[['session_id', 'level_group', 'index2', 'elapsed_time']],\n                          left_on = ['session_id', 'level_group', 'index'],\n                          right_on = ['session_id', 'level_group', 'index2'],\n                          suffixes = (\"_curr\", \"_prev\"))\n\ntrain_df['time_spent'] = train_df['elapsed_time_curr'] - train_df['elapsed_time_prev']\ntrain_df.drop(columns=['elapsed_time_curr', 'elapsed_time_prev', 'index2_curr', 'index2_prev'], inplace=True)\n\ntrain_df['time_spent'] = np.abs(train_df['time_spent'])\ntrain_df['time_spent'] = train_df['time_spent'].clip(0, 10)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:36:09.916606Z","iopub.execute_input":"2023-04-06T12:36:09.917500Z","iopub.status.idle":"2023-04-06T12:38:13.736539Z","shell.execute_reply.started":"2023-04-06T12:36:09.917459Z","shell.execute_reply":"2023-04-06T12:38:13.735636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:13.738046Z","iopub.execute_input":"2023-04-06T12:38:13.739091Z","iopub.status.idle":"2023-04-06T12:38:13.754090Z","shell.execute_reply.started":"2023-04-06T12:38:13.739054Z","shell.execute_reply":"2023-04-06T12:38:13.752743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_group_level(q):\n    qno = int(q[1:])\n    if qno < 4:\n        return '0-4'\n    elif qno < 14:\n        return '5-12'\n    return '13-22'","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:13.755325Z","iopub.execute_input":"2023-04-06T12:38:13.755787Z","iopub.status.idle":"2023-04-06T12:38:13.766004Z","shell.execute_reply.started":"2023-04-06T12:38:13.755740Z","shell.execute_reply":"2023-04-06T12:38:13.764786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\ntrain_label['q'] = train_label['session_id'].apply(lambda s: s.split(\"_\")[-1])\ntrain_label['session_id'] = train_label['session_id'].apply(lambda s: int(s.split(\"_\")[0]))\ntrain_label['level_group'] = train_label.q.apply(get_group_level)\n\ntrain_label.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:13.768616Z","iopub.execute_input":"2023-04-06T12:38:13.769158Z","iopub.status.idle":"2023-04-06T12:38:14.964687Z","shell.execute_reply.started":"2023-04-06T12:38:13.769101Z","shell.execute_reply":"2023-04-06T12:38:14.963670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# elapsed time stats","metadata":{}},{"cell_type":"code","source":"def get_elapsed_stats(df):\n    level_stat = df.groupby(['session_id', 'level'])[['time_spent']].sum().reset_index()\n    group_stat = df.groupby(['session_id', 'level_group'])[['time_spent']].sum().reset_index()\n    gp_event_stat = df.groupby(['session_id', 'level_group', 'event_name'])[['time_spent']].sum().reset_index()\n    \n    for l in level_stat.level.unique():\n        arr = level_stat[level_stat.level == l]['time_spent'].values\n        mx = np.quantile(arr, 0.99)\n        print(\"Level:{} -->{}\".format(l, mx))\n        print(\"max: \", np.max(arr))\n        level_stat.loc[level_stat['level'] == l, 'time_spent'] = np.clip(arr, 0, mx)\n    \n    print()\n    print()\n    \n    for g in group_stat.level_group.unique():\n        arr = group_stat[group_stat.level_group == g]['time_spent'].values\n        mx = np.quantile(arr, 0.99)\n        print(\"group:{} -->{}\".format(g, mx))\n        print(\"max: \", np.max(arr))\n        group_stat.loc[group_stat['level_group'] == g, 'time_spent'] = np.clip(arr, 0, mx)\n    \n    print()\n    print()\n    \n    for g in gp_event_stat.event_name.unique():\n        arr = gp_event_stat[gp_event_stat.event_name == g]['time_spent'].values\n        mx = np.quantile(arr, 0.99)\n        print(\"event:{} -->{}\".format(g, mx))\n        print(\"max: \", np.max(arr))\n        gp_event_stat.loc[gp_event_stat['event_name'] == g, 'time_spent'] = np.clip(arr, 0, mx)\n    \n    \n    return {\n        'level_stat' : level_stat,\n        'group_stat': group_stat,\n        'gp_event_stat': gp_event_stat\n    }","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:14.966056Z","iopub.execute_input":"2023-04-06T12:38:14.966651Z","iopub.status.idle":"2023-04-06T12:38:14.978813Z","shell.execute_reply.started":"2023-04-06T12:38:14.966614Z","shell.execute_reply":"2023-04-06T12:38:14.977632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nstat_map = get_elapsed_stats(train_df)\nlevel_stat = stat_map['level_stat']\ngroup_stat = stat_map['group_stat']\ngp_event_stat = stat_map['gp_event_stat']","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:14.980382Z","iopub.execute_input":"2023-04-06T12:38:14.980712Z","iopub.status.idle":"2023-04-06T12:38:26.611902Z","shell.execute_reply.started":"2023-04-06T12:38:14.980682Z","shell.execute_reply":"2023-04-06T12:38:26.610622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 5))\nfig.suptitle(\"time spent per level group\")\n\nfor i, lg in enumerate(['0-4', '5-12', '13-22']):\n    arr = group_stat[group_stat.level_group == lg]['time_spent'].values\n    mx = np.quantile(arr, 0.995)\n    arr = np.clip(arr, 0, mx)\n    \n    sns.histplot(arr, stat='density', ax=ax[0], label=lg)\n    sns.histplot(np.log(1+arr), stat='density', ax=ax[1], label=lg)\n    \nax[0].set_title(\"elapsed time\")\nax[1].set_title(\"log elapsed time\")\n\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:26.613564Z","iopub.execute_input":"2023-04-06T12:38:26.614006Z","iopub.status.idle":"2023-04-06T12:38:28.173851Z","shell.execute_reply.started":"2023-04-06T12:38:26.613971Z","shell.execute_reply":"2023-04-06T12:38:28.172473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"event_lst = list(train_df.event_name.unique())\nfor event_name in event_lst:\n    for i, lg in enumerate(['0-4', '5-12', '13-22']):\n        event_df = gp_event_stat[(gp_event_stat.level_group == lg) &  \n                                 (gp_event_stat.event_name == event_name)]\n\n        sns.histplot(x=np.log(1+event_df.time_spent), label=lg)\n    plt.title(event_name)\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:28.175417Z","iopub.execute_input":"2023-04-06T12:38:28.175878Z","iopub.status.idle":"2023-04-06T12:38:47.136770Z","shell.execute_reply.started":"2023-04-06T12:38:28.175839Z","shell.execute_reply":"2023-04-06T12:38:47.135453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations**\n\n1. In most of the clicks, group(13-22) has more time spent relative to other groups.\n2. Time spent on Notebook has is skewed to right, with more elongated distribution for 13-22\n3. time spent on the navigate click increased with level group","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, lg in enumerate(['0-4', '5-12', '13-22']):\n    navigate_df = gp_event_stat[(gp_event_stat.level_group == lg) & \n                                (gp_event_stat.event_name == 'navigate_click')]\n    reset_df = gp_event_stat[(gp_event_stat.level_group == lg) & \n                                (gp_event_stat.event_name != 'navigate_click')]\n    reset_df = reset_df.groupby('session_id')[['time_spent']].sum().reset_index()\n    \n    \n    \n    fig, ax = plt.subplots(1, 2, figsize=(12, 5))\n    \n    if i == 0:\n        fig.suptitle(\"distribution of time spent - Navigate (vs) other events\")\n    \n    sns.histplot(data=navigate_df, x='time_spent', label=\"navigate click\", ax=ax[0])\n    plt.xticks(rotation=45)\n    \n    sns.histplot(data=reset_df, x='time_spent', label=\"other clicks\", ax=ax[0])\n    plt.xticks(rotation=45)\n    \n    \n    \n    sns.histplot(x=np.log(1+navigate_df['time_spent']), label=\"navigate click\", ax=ax[1])\n    plt.xticks(rotation=45)\n    \n    sns.histplot(x=np.log(1+reset_df['time_spent']),  label=\"other clicks\", ax=ax[1])\n    plt.xticks(rotation=45)\n    \n    \n    ax[0].set_title(lg)\n    ax[1].set_title(\"log({})\".format(lg))\n    \n    \n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:47.138201Z","iopub.execute_input":"2023-04-06T12:38:47.138631Z","iopub.status.idle":"2023-04-06T12:38:52.314029Z","shell.execute_reply.started":"2023-04-06T12:38:47.138596Z","shell.execute_reply":"2023-04-06T12:38:52.312914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observation**\n1. for the navigate event --> avg time spent for level group(0-4) is more than the other events.\n2. for level group(5-12, 13-22) --> avg time spent is on navigate is less than other events.\n3. This difference between the groups may be due to, at the begining of the game, user is exploring the navigation and may not be sure that ","metadata":{}},{"cell_type":"markdown","source":"# Distribution of wrong answers","metadata":{"execution":{"iopub.status.busy":"2023-04-06T05:18:40.392893Z","iopub.execute_input":"2023-04-06T05:18:40.393660Z","iopub.status.idle":"2023-04-06T05:18:40.480313Z","shell.execute_reply.started":"2023-04-06T05:18:40.393616Z","shell.execute_reply":"2023-04-06T05:18:40.479268Z"}}},{"cell_type":"code","source":"train_label['wrong'] = 1-train_label.correct\nwrong_df = train_label.groupby(['session_id', 'level_group'])[['wrong']].sum().reset_index()\n\nfor lg in ['0-4', '5-12', '13-22']:\n    tmp_df = wrong_df[wrong_df.level_group == lg]\n    \n    plt.figure(figsize=(8, 4))\n    plt.title(lg)\n    sns.countplot(data=tmp_df, x='wrong')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:52.317227Z","iopub.execute_input":"2023-04-06T12:38:52.317676Z","iopub.status.idle":"2023-04-06T12:38:52.924568Z","shell.execute_reply.started":"2023-04-06T12:38:52.317641Z","shell.execute_reply":"2023-04-06T12:38:52.923701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# how does the distribution changes for the correct and wrong answers with time ?","metadata":{}},{"cell_type":"code","source":"def plot_time_spent_plot(df, lg, ax):\n    if lg == '0-4':\n        mn_wrong = 0; mx_wrong=1\n    \n    if lg == '5-12':\n        mn_wrong = 0; mx_wrong=6\n    \n    if lg == '13-22':\n        mn_wrong = 0; mx_wrong = 2\n    \n    \n    correct_sessids = wrong_df[(wrong_df.level_group == lg) & \n                             (wrong_df.wrong <= mn_wrong)]['session_id'].unique()\n    \n    wrong_sessids = wrong_df[(wrong_df.level_group == lg) & \n                             (wrong_df.wrong >= mx_wrong)]['session_id'].unique()\n    \n    \n    tmp_df = df[df.level_group == lg]\n    sns.histplot(x=np.log(1+tmp_df[tmp_df.session_id.isin(wrong_sessids)]['time_spent']), \n                 stat='density', \n                 label='wrong',\n                 ax=ax)\n    \n    sns.histplot(x=np.log(1+tmp_df[tmp_df.session_id.isin(correct_sessids)]['time_spent']), \n                 stat='density', \n                 label='correct',\n                 ax=ax)\n    \n    ax.set_title(lg)","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:52.925691Z","iopub.execute_input":"2023-04-06T12:38:52.926209Z","iopub.status.idle":"2023-04-06T12:38:52.935013Z","shell.execute_reply.started":"2023-04-06T12:38:52.926176Z","shell.execute_reply":"2023-04-06T12:38:52.934085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-06T10:44:27.839406Z","iopub.execute_input":"2023-04-06T10:44:27.840247Z","iopub.status.idle":"2023-04-06T10:44:27.846324Z","shell.execute_reply.started":"2023-04-06T10:44:27.840205Z","shell.execute_reply":"2023-04-06T10:44:27.845198Z"}}},{"cell_type":"code","source":"def plot_group_level_elapsed():\n    fig, ax = plt.subplots(1, 3, figsize=(12, 5))\n    fig.suptitle(\"time spent distribution for level group\")\n    for i, lg in enumerate(['0-4', '5-12', '13-22']):\n        plot_time_spent_plot(group_stat, lg, ax[i])\n    plt.legend()\n    plt.show()\nplot_group_level_elapsed()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:52.936490Z","iopub.execute_input":"2023-04-06T12:38:52.937386Z","iopub.status.idle":"2023-04-06T12:38:54.035469Z","shell.execute_reply.started":"2023-04-06T12:38:52.937349Z","shell.execute_reply":"2023-04-06T12:38:54.034345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_group_level_elapsed():\n    fig, ax = plt.subplots(1, 3, figsize=(12, 5))\n    fig.suptitle(\"time spent distribution for level group\")\n    for i, lg in enumerate(['0-4', '5-12', '13-22']):\n        plot_time_spent_plot(group_stat, lg, ax[i])\n    plt.show()\n    \nplot_group_level_elapsed()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:54.036866Z","iopub.execute_input":"2023-04-06T12:38:54.037172Z","iopub.status.idle":"2023-04-06T12:38:55.060283Z","shell.execute_reply.started":"2023-04-06T12:38:54.037143Z","shell.execute_reply":"2023-04-06T12:38:55.059180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_event_elapsed(event_name):\n    fig, ax = plt.subplots(1, 3, figsize=(12, 5))\n    fig.suptitle(\"time spent distribution for event \"+ event_name)\n    for i, lg in enumerate(['0-4', '5-12', '13-22']):\n        plot_time_spent_plot(gp_event_stat[gp_event_stat.event_name == event_name], lg, ax[i])\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:55.061702Z","iopub.execute_input":"2023-04-06T12:38:55.062240Z","iopub.status.idle":"2023-04-06T12:38:55.071474Z","shell.execute_reply.started":"2023-04-06T12:38:55.062203Z","shell.execute_reply":"2023-04-06T12:38:55.069925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for event_name in list(gp_event_stat.event_name.unique()):\n    if event_name == 'checkpoint':\n        continue\n    \n    plot_event_elapsed(event_name)","metadata":{"execution":{"iopub.status.busy":"2023-04-06T12:38:55.073466Z","iopub.execute_input":"2023-04-06T12:38:55.073863Z","iopub.status.idle":"2023-04-06T12:39:11.219462Z","shell.execute_reply.started":"2023-04-06T12:38:55.073826Z","shell.execute_reply":"2023-04-06T12:39:11.218175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Hyposthesis:**\n1. cutscene, person click are more important factors.Between the right and wrong answeres the distribution is high.\n2. Navigate click; this is not that important as it seems doesn't induce any knowledge.\n\n\n**Observations:**\n1. Person click distribution for level group is same both the correct and wrong answer sessions\n2. Interestingly more deviation is shown in navigation_click, object_click,object_hover,observation_click , map_click, map_hover.\n\n3. Histogram representations showed that my hypothesis on time spent distribution on person click, cutscene and navigate clicks are wrong.\n\n4. The reason i think is that cutscene, person click are the mandatory that every student has to go through, where as \nfor quick navigation, any kid has to properly understand the dialogs. \n\n  eg. in the level group(0-4), to navigate to gradpa's room player has to understand, which direction to click for navigation. if went wrong (or) more time spent, chances that click person dialogs multiple times if exists on screen.","metadata":{}}]}