{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')\nex_submission = pd.read_csv('../input/riiid-test-answer-prediction/example_sample_submission.csv')\nex_test = pd.read_csv('../input/riiid-test-answer-prediction/example_test.csv')\nquestions = pd.read_csv('../input/riiid-test-answer-prediction/questions.csv')\ntrain = pd.read_csv('../input/riiid-test-answer-prediction/train.csv', nrows = 99999)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"Remove the content_type_id is 1"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.drop(train[train['content_type_id']==1].index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Change name content_id into question_id, so we can merge with questions dataframe"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.rename(columns={'content_id':'question_id'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Merge train and questions dataframe"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.merge(questions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"User_id with most questions answered."},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train,\n                   x='user_id', \n                   histfunc='count',\n                  )\nfig.update_layout(showlegend =False,\n                 width=1000,\n                 height=600)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"User_id with longest time stamp"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train,\n                   x='user_id', \n                   y='timestamp',\n                   histfunc='max',\n                  )\nfig.update_layout(showlegend =False,\n                 width=1000,\n                 height=600)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Average of answered_correctly each user"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train,\n                   x='user_id', \n                   y='answered_correctly',\n                   histfunc='avg',\n                  )\nfig.update_layout(showlegend =False,\n                 width=1000,\n                 height=600)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"avg of answered correctly each question"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train,\n                   x='question_id', \n                   y='answered_correctly',\n                   histfunc='avg',\n                  )\nfig.update_layout(showlegend =False,\n                 width=1000,\n                 height=600)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"avg of answered correctly each part"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram(train,\n                   x='part', \n                   y='answered_correctly',\n                   histfunc='avg',\n                  )\nfig.update_layout(showlegend =False,\n                 width=1000,\n                 height=600)\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Check the correlation with answered_correctly"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\nsns.barplot(x='index', y ='answered_correctly', data=train.corr().drop('answered_correctly').reset_index())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"all is bad, no one is more than 0.1"},{"metadata":{},"cell_type":"markdown","source":"For now I will drop row_id, content_type_id, user_answer, correct_answer.<br>\nI think those are no corr even in real life.<br>\nI mean we don't care correct answer, is it 1 or 2 or 3"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.drop(['row_id', 'content_type_id', 'user_answer', 'correct_answer'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"How do you define the question is hard or easy?<br>\nYap!, by count average of how many answered correctly in each question.<br>\nSo. we gonna groupby question_id and see the average of answered_correctly."},{"metadata":{"trusted":true},"cell_type":"code","source":"question_answered_correctly = train.groupby('question_id').mean()['answered_correctly'].reset_index()\nquestion_answered_correctly","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We gonna merge them with train data frame and name it as PQAC (percentage of questions answered correctly)"},{"metadata":{"trusted":true},"cell_type":"code","source":"question_answered_correctly = question_answered_correctly.rename(columns={'answered_correctly':'PQAC'})\ntrain = train.merge(question_answered_correctly)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Next, they already divided question into part. <br>\nLet's which question is hard by part."},{"metadata":{"trusted":true},"cell_type":"code","source":"part_answered_correctly = train.groupby('part').mean()['answered_correctly'].reset_index()\npart_answered_correctly","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Again, We merge with our train dataframe. As PPQAC (percentage of part questions answered correctly)"},{"metadata":{"trusted":true},"cell_type":"code","source":"part_answered_correctly = part_answered_correctly.rename(columns={'answered_correctly':'PPQAC'})\ntrain = train.merge(part_answered_correctly)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"How do you define user is smart or not? <br>\nMaybe they give many correct answer"},{"metadata":{"trusted":true},"cell_type":"code","source":"user_answered_correctly = train.groupby('user_id').mean()['answered_correctly'].reset_index()\nuser_answered_correctly","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Again, We merge with our train dataframe. As UAC (percentage of user answered correctly)"},{"metadata":{"trusted":true},"cell_type":"code","source":"user_answered_correctly = user_answered_correctly.rename(columns={'answered_correctly':'UAC'})\ntrain = train.merge(user_answered_correctly)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"how about average user to answer?"},{"metadata":{"trusted":true},"cell_type":"code","source":"avg_user_answered = (train.groupby('user_id').max()['timestamp']/train.groupby('user_id').count()['timestamp']).reset_index()\navg_user_answered","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We merge with our train dataframe. As AUA (Average User Answer)"},{"metadata":{"trusted":true},"cell_type":"code","source":"avg_user_answered = avg_user_answered.rename(columns={'timestamp':'AUA'})\ntrain = train.merge(avg_user_answered)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"But quick to answer question is not enough for people to be called smart. <br>\nSo I multiple (percentage of user answered correctly and Average User Answer <br>\nAnd call it AUAC (Average User Answered Correctly)"},{"metadata":{"trusted":true},"cell_type":"code","source":"train['AUAC'] = train['UAC']*train['AUA']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Check again the correlation"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\nsns.barplot(x='index', y ='answered_correctly', data=train.corr().drop('answered_correctly').reset_index())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from datetime import datetime","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['datetime'] = pd.to_datetime(train['timestamp'], unit='ms')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.index = train['datetime']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"z = train.groupby('user_id').resample('1D').count()[0:20]['answered_correctly'].reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"z = z.rename(columns={'answered_correctly':'question/day', 'datetime': 'date_id'})\nz","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['date_id'] = train['datetime'].dt.date","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[20:40]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.scatterplot(x='timestamp', y='answered_correctly',data=train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.line(train,\n              x='timestamp', \n              y='answered_correctly',\n              color='user_id',\n              #text='content_id',\n                 #symbol='Outlet_Type',\n                 #text='Outlet_Identifier',\n             )#.update_yaxes(categoryorder='total ascending')\n\nfig.update_traces(marker=dict(size=12,),\n                  textposition='top center',\n                  textfont=dict(family='Arial',size=12),\n              \n                 )\nfig.update_layout(\n    height=600,\n)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}