{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport collections","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-11T15:55:15.479444Z","iopub.execute_input":"2023-04-11T15:55:15.481791Z","iopub.status.idle":"2023-04-11T15:55:15.524676Z","shell.execute_reply.started":"2023-04-11T15:55:15.481710Z","shell.execute_reply":"2023-04-11T15:55:15.523320Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/input/predict-student-performance-from-game-play/train.csv'\nsession_id = pd.read_csv(train_path, usecols = [\"session_id\"])\nlevel = pd.read_csv(train_path, usecols = [\"level\"])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-11T15:55:19.415575Z","iopub.execute_input":"2023-04-11T15:55:19.415965Z","iopub.status.idle":"2023-04-11T15:57:17.957263Z","shell.execute_reply.started":"2023-04-11T15:55:19.415930Z","shell.execute_reply":"2023-04-11T15:57:17.955874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- In the [discuttion][2], it is pointed out that levels 7, 15, 20, 21 and 22 are skipped.\n\n[2]: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/390339#2174184\n\n- As other [problems][1] with the dataset have been noted, the dataset has recently been added. We analyzed the new dataset to see if the problem of skipping levels has been resolved.\n\n[1]: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/395250\n\n- From the following analysis, we see that all sessions has ```level==22```. However, some sessions lacks levels 7, 15, 20, 21. \n\n- It was noted that there are several columns like index and elapsed_time in this dataset that do not seem to be consistent, and the competition host indicated that this may be due to a bug in the game.  Since only certain levels are missing, perhaps this missingness is a bug of the game.\n\n- Please let me know if there is already another similar analysis. I'm new to kaggle and this is my first competition so please tell me something to improve this notebook and how to use kaggle.","metadata":{}},{"cell_type":"code","source":"nlevel = pd.concat([session_id, level], axis=1).groupby('session_id').nunique()\nprint(nlevel[nlevel.level<21])\nprint('\\nNumber of unique levels are larger than 20 for all sessions.')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-11T15:58:05.031029Z","iopub.execute_input":"2023-04-11T15:58:05.031449Z","iopub.status.idle":"2023-04-11T15:58:08.272930Z","shell.execute_reply.started":"2023-04-11T15:58:05.031413Z","shell.execute_reply":"2023-04-11T15:58:08.271871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n23 = len(nlevel[nlevel.level==23])\nn22 = len(nlevel[nlevel.level==22])\nn21 = len(nlevel[nlevel.level==21])\nlen_ = len(nlevel)\n\nprint('0 missing:'.ljust(15) + f'{n23: 10d} sessions, {(n23/len_)*100: 10.3f} %')\nprint('1 missing:'.ljust(15) + f'{n22: 10d} sessions, {(n22/len_)*100: 10.3f} %')\nprint('2 missing:'.ljust(15) + f'{n21: 10d} sessions, {(n21/len_)*100: 10.3f} %')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-11T15:58:10.154056Z","iopub.execute_input":"2023-04-11T15:58:10.154804Z","iopub.status.idle":"2023-04-11T15:58:10.165103Z","shell.execute_reply.started":"2023-04-11T15:58:10.154748Z","shell.execute_reply":"2023-04-11T15:58:10.164014Z"},"jupyter":{"outputs_hidden":true,"source_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_and_level = pd.concat([session_id, level], axis=1)\nfull_levels = set(np.arange(0, 23))\none_missing_session = nlevel[nlevel.level==22].index\ndiffs = []\nfor ID in one_missing_session:\n    tmp = session_and_level.loc[session_and_level.session_id == ID]\n    diff = full_levels - set(tmp.level.unique())\n    diffs.append(diff)\ntmp = []\nfor diff in diffs:\n    for e in diff:\n        tmp.append(e)\nprint(f'Missing values distribution: {collections.Counter(tmp)} in {n22} sessions')\n\ntwo_missing_session = nlevel[nlevel.level==21].index\ndiffs = []\nfor ID in two_missing_session:\n    tmp = session_and_level.loc[session_and_level.session_id == ID]\n    diff = full_levels - set(tmp.level.unique())\n    diffs.append(diff)\ntmp = []\nfor diff in diffs:\n    for e in diff:\n        tmp.append(e)\nprint(f'Missing values distribution: {collections.Counter(tmp)} in {n21} sessions')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-04-11T15:58:21.567680Z","iopub.execute_input":"2023-04-11T15:58:21.568105Z","iopub.status.idle":"2023-04-11T15:58:39.951405Z","shell.execute_reply.started":"2023-04-11T15:58:21.568068Z","shell.execute_reply":"2023-04-11T15:58:39.950272Z"},"trusted":true},"execution_count":null,"outputs":[]}]}