{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nimport plotly.express as px\nfrom collections import Counter as count\n\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.layers import Input, Dense\nfrom sklearn.model_selection import KFold\n\n\ndebug = False","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Fork from  https://www.kaggle.com/piantic/riiid-answer-correctness-prediction-basic-eda"},{"metadata":{"trusted":true},"cell_type":"code","source":"\nif debug:\n    read_num = 10**6\nelse:\n    read_num = 10**7\n\ntrain = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', low_memory=False, nrows=read_num, \n                       dtype={'row_id': 'int64', 'timestamp': 'int64', 'user_id': 'int32', 'content_id': 'int16', 'content_type_id': 'int8',\n                              'task_container_id': 'int16', 'user_answer': 'int8', 'answered_correctly': 'int8', 'prior_question_elapsed_time': 'float32', \n                             'prior_question_had_explanation': 'boolean',\n                             }\n                      )\ntest = pd.read_csv('../input/riiid-test-answer-prediction/example_test.csv')\nsubmit = pd.read_csv('../input/riiid-test-answer-prediction/example_sample_submission.csv')\n\nprint('Train shapes: ', train.shape)\nprint('Test shapes: ', test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submit","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Fork form https://www.kaggle.com/isaienkov/openvaccine-eda-feature-engineering-modeling"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"fig = px.histogram(\n    train, \n    \"task_container_id\", \n    nbins=25, \n    title='task_container_id column distribution', \n    width=700,\n    height=500\n)\nfig.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(\n    train, \n    \"timestamp\", \n    nbins=25, \n    title='timestamp column distribution', \n    width=700,\n    height=500\n)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(\n    train, \n    \"prior_question_elapsed_time\", \n    nbins=25, \n    title='prior_question_elapsed_time column distribution', \n    width=700,\n    height=500\n)\nfig.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ds = train['answered_correctly'].value_counts().reset_index()\nds.columns = ['answered_correctly', 'count']\nfig = px.pie(\n    ds, \n    values='count', \n    names=\"answered_correctly\", \n    title='answered_correctly bar chart', \n    width=500, \n    height=500\n)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ds = train['prior_question_had_explanation'].value_counts().reset_index()\nds.columns = ['prior_question_had_explanation', 'count']\nfig = px.pie(\n    ds, \n    values='count', \n    names=\"prior_question_had_explanation\", \n    title='prior_question_had_explanation bar chart', \n    width=500, \n    height=500\n)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nds = train['user_answer'].value_counts().reset_index()\nds.columns = ['user_answer', 'count']\nfig = px.pie(\n    ds, \n    values='count', \n    names=\"user_answer\", \n    title='user_answer bar chart', \n    width=500, \n    height=500\n)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nds = train['content_type_id'].value_counts().reset_index()\nds.columns = ['content_type_id', 'count']\nfig = px.pie(\n    ds, \n    values='count', \n    names=\"content_type_id\", \n    title='content_type_id bar chart', \n    width=500, \n    height=500\n)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}