{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Accuracy for each question, session\nThe label dataset was used to examine the percentage of correct answers for each question, session.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:20:48.937533Z","iopub.execute_input":"2023-06-08T23:20:48.938169Z","iopub.status.idle":"2023-06-08T23:20:48.966469Z","shell.execute_reply.started":"2023-06-08T23:20:48.938132Z","shell.execute_reply":"2023-06-08T23:20:48.965569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv(\n    '/kaggle/input/predict-student-performance-from-game-play/train_labels.csv'\n)\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:20:48.968120Z","iopub.execute_input":"2023-06-08T23:20:48.968687Z","iopub.status.idle":"2023-06-08T23:20:50.589061Z","shell.execute_reply.started":"2023-06-08T23:20:48.968637Z","shell.execute_reply":"2023-06-08T23:20:50.587984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Total Accuracy\ntotal_acc = targets['correct'].mean()\nprint(f'Total accuracy: {total_acc}')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:20:50.592812Z","iopub.execute_input":"2023-06-08T23:20:50.594882Z","iopub.status.idle":"2023-06-08T23:20:50.601521Z","shell.execute_reply.started":"2023-06-08T23:20:50.594844Z","shell.execute_reply":"2023-06-08T23:20:50.600688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Accuracy for each question","metadata":{}},{"cell_type":"code","source":"qs = []\nacc_q = []\nfor q_num in range(18):\n    label_0 = 0\n    label_1 = 0\n    for _, row in targets.loc[targets.q == (q_num + 1)].iterrows():\n        if row['correct'] == 0:\n            label_0 += 1\n        elif row['correct'] == 1:\n            label_1 += 1\n    \n    acc = label_1 / (label_0 + label_1)\n    \n    # For bar graph\n    qs.append(q_num + 1)\n    acc_q.append(acc)\n    \n    print('='*30)\n    print(f'Question: {q_num + 1}')\n    print(f'label=0 count: {label_0}')\n    print(f'label=1 count: {label_1}')\n    print(f'Accuracy: {acc}')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:20:50.607093Z","iopub.execute_input":"2023-06-08T23:20:50.607823Z","iopub.status.idle":"2023-06-08T23:21:17.511475Z","shell.execute_reply.started":"2023-06-08T23:20:50.607792Z","shell.execute_reply":"2023-06-08T23:21:17.510306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bar graph of accuracy for each question\nplt.bar(qs, acc_q, tick_label=qs, align='center')\nplt.ylabel('Accuracy for each question')\nplt.xlabel('Qusestion')  \nplt.axhline(y=total_acc, color='orange', linestyle='dashed') \nplt.axhline(y=0.5, color='purple', linestyle='dashed') \nplt.grid(color = 'gray', linestyle='--')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:24:01.526274Z","iopub.execute_input":"2023-06-08T23:24:01.526717Z","iopub.status.idle":"2023-06-08T23:24:01.762440Z","shell.execute_reply.started":"2023-06-08T23:24:01.526680Z","shell.execute_reply":"2023-06-08T23:24:01.761436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Accuracy for each session","metadata":{}},{"cell_type":"code","source":"asc_df = targets.groupby('session')['correct'].mean().sort_values()\nprint(f'Worst TOP 100 accuracy: {asc_df.head(50)}')\nprint('='*50)\n\ndesc_df = targets.groupby('session')['correct'].mean().sort_values(ascending=False)\nprint(f'Best TOP 100 accuracy: {desc_df.head(50)}')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:21:17.779630Z","iopub.execute_input":"2023-06-08T23:21:17.779939Z","iopub.status.idle":"2023-06-08T23:21:17.841337Z","shell.execute_reply.started":"2023-06-08T23:21:17.779909Z","shell.execute_reply":"2023-06-08T23:21:17.840110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram of accuracy for each session\nplt.hist(desc_df, bins=18)\nplt.axvline(x=total_acc, color='orange', linestyle='dashed') \nplt.axvline(x=0.5, color='purple', linestyle='dashed') \nplt.xlabel('Accuracy for each session')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:23:52.391899Z","iopub.execute_input":"2023-06-08T23:23:52.392263Z","iopub.status.idle":"2023-06-08T23:23:52.566663Z","shell.execute_reply.started":"2023-06-08T23:23:52.392230Z","shell.execute_reply":"2023-06-08T23:23:52.565692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}