{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\n\nlectures_types = {\n    'lecture_id': 'int16',\n    'tag': 'int16',\n    'part': 'int8',\n    'type_of': 'string'\n}\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv', dtype = lectures_types) # wszystkie dane ładowane do pamięci\nlectures.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_types = {\n    'row_id': 'int64',\n    'timestamp': 'int64',\n    'user_id': 'int32',\n    'content_id': 'int16',\n    'content_type_id': 'int8',\n    'task_container_id': 'int16',\n    'user_answer': 'int8',\n    'answered_correctly': 'int8',\n    'prior_question_elapsed_time': 'float32', \n    'prior_question_had_explanation': 'boolean'\n}        \n\n\ntrain = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', dtype = train_types) # wszystkie dane ładowane do pamięci\n# train = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', dtype = train_types, nrows=1000) # pierwsze 1000 wierszy","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"print('W zbiorze istnieje ' + str(lectures.shape[0]) + ' wykładów')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"types = lectures.type_of.value_counts()\n\nfig = plt.figure(figsize=(13,5))\nplt.bar(types.index, types.values, color='orange')\nplt.xlabel(\"Typy wykładków\")\nplt.ylabel(\"Ilość wystąpień w zbiorze\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_lect = train.groupby([\"user_id\", \"answered_correctly\"]).size().unstack()\nuser_lect.columns = ['Wyklad', 'Zle', 'Poprawne']\nuser_lect['Wyklad'] = user_lect['Wyklad'].fillna(0)\nuser_lect = user_lect.astype('Int64')\nuser_lect['Wyklad zaliczony'] = np.where(user_lect['Wyklad'] > 0, True, False)\n\nwatches_l = user_lect.groupby(\"Wyklad zaliczony\").agg({'Zle': ['sum'], 'Poprawne': ['sum']})\n\nwatches_l['Procent poprawności'] = watches_l['Poprawne'] / (watches_l['Poprawne'] + watches_l['Zle'])\n\nwatches_l = watches_l.iloc[:,2]\n\nfig = plt.figure(figsize=(13, 5))\nwatches_l.plot.bar(color=['red', 'lightgreen'])\n\nplt.xlabel(\"Użytkowanik brał udział w przynajmniej jednym wykładzie\")\nplt.ylabel(\"Procent poprawności odpowiedzi\")\nplt.xticks(x, labels=['Nie', 'Tak'], rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"task_containers = train.groupby([\"task_container_id\", \"answered_correctly\"]).size().unstack()\ntask_containers.columns = ['Wyklad', 'Zle', 'Poprawne']\ntask_containers = task_containers.astype('Int64')\ntask_containers['Procent poprawności'] = task_containers['Poprawne'] / (task_containers['Poprawne'] + task_containers['Zle'])\ntask_containers['Jaki procent wykladow'] = task_containers['Wyklad'] / (task_containers['Wyklad'] + task_containers['Zle'] + task_containers['Poprawne'])\ntask_containers = task_containers.fillna(0)\ntask_containers[['Wyklad', 'Procent poprawności', 'Jaki procent wykladow']].sort_values(by=\"Jaki procent wykladow\", ascending=False).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"task_containers[['Wyklad', 'Procent poprawności', 'Jaki procent wykladow']].sort_values(by=\"Wyklad\", ascending=False).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"task_containers['Byl wyklad'] = np.where(task_containers['Wyklad'] == 0, False, True)\ntmp = task_containers.groupby(\"Byl wyklad\").sum()\ntmp['Procent poprawności'] = tmp['Poprawne'] / (tmp['Poprawne'] + tmp['Zle'])\ntmp = tmp['Procent poprawności']\n\nx = list(tmp.index)\ny = list(tmp.values)\n\nfig = plt.figure(figsize = (13, 5)) \nplt.bar(x, y, color=['red', 'lightgreen'])\n  \nplt.xlabel('Czy w danym task_container był wykład') \nplt.ylabel('Procent poprawności') \nplt.xticks(x, labels=['Nie', 'Tak'])\nplt.show() ","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}