{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import dask.dataframe as dd\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Lectures Dataset\n\n\n\nlectures.csv: metadata for the lectures watched by users as they progress in their education.\n\n    lecture_id: foreign key for the train/test content_id column, when the content type is lecture (1).\n\n    part: top level category code for the lecture.\n\n    tag: one tag codes for the lecture. The meaning of the tags will not be provided, but these codes are sufficient for clustering the lectures together.\n\n    type_of: brief description of the core purpose of the lecture\n\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"lect=pd.read_csv (\"/kaggle/input/riiid-test-answer-prediction/lectures.csv\")\nlect","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"part_count = lect.groupby(\"part\")['lecture_id'].count().reset_index(name = 'counts')\nchart = px.pie(part_count, values='counts', names='part', title='Part Type ')\nchart.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type_count = lect.groupby(\"type_of\")['lecture_id'].count().reset_index(name = 'counts')\nchart2 = px.pie(type_count, values='counts', names='type_of', title='Lecture Type')\nchart2.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Questions Dataset\n\n\n\nquestions.csv: metadata for the questions posed to users.\n\n    question_id: foreign key for the train/test content_id column, when the content type is question (0).\n\n    bundle_id: code for which questions are served together.\n\n    correct_answer: the answer to the question. Can be compared with the train user_answer column to check if the user was right.\n\n    part: top level category code for the question.\n\n    tags: one or more detailed tag codes for the question. The meaning of the tags will not be provided, but these codes are sufficient for clustering the questions together.\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"ques=pd.read_csv (\"/kaggle/input/riiid-test-answer-prediction/questions.csv\")\nques","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(8,6))\ncr = ques.groupby(\"correct_answer\")['question_id'].count().reset_index(name = 'counts')\ncad = sns.barplot(x=\"correct_answer\", y=\"counts\", data=cr).set_title(\"Correct Answers Distribution\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"check = ques['tags'].str.split(' ').explode('tags').reset_index()\ncheck = check['tags'].value_counts().reset_index()\n\ncheck.columns = ['tag', 'count']\ncheck['tag'] = check['tag'].astype(str) + '-'\ncheck = check.sort_values(['count'])\n\nfig = px.bar(\n    check.tail(10), \n    x='count', \n    y='tag', \n    title='Top 10 most useful tags'\n)\nfig.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}