{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install datatable==0.11.0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import datatable as dt\n\ntrain = dt.fread(\"../input/riiid-test-answer-prediction/train.csv\").to_pandas()\n\nprint(train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# loading other data files\nimport pandas as pd\nquestions_df = pd.read_csv('../input/riiid-test-answer-prediction/questions.csv')\nlectures_df = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')\ntest_df = pd.read_csv('../input/riiid-test-answer-prediction/example_test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"P_slip : learned concept previously but forgot the answer -> look for identical questions that were wrong \n\n-> assume that this probability decreases over time\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# time series of prior qn elapsed time?\ntrain[train['user_id']==115]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user968220408 = train[train['user_id']==968220408]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user968220408","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(user968220408)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nxt = np.arange(208)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"yt =user968220408['prior_question_elapsed_time']/1000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"yt1 = user968220408['answered_correctly']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load question difficulty\nimport pandas as pd\ntags_df = pd.read_csv(\"../input/tagscsv/mytags1.csv\")\n#tagscsv.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"difficultyweight = 1 -tags_df['Percent_correct']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#difficultyweight.head()\ntags_df['diffweight'] = difficultyweight","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t1 = tags_df[tags_df['tag']==24]\nt2 = t1['Percent_correct']\nt2[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train1 = train[:100000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"\nmerge with qns"},{"metadata":{"trusted":true},"cell_type":"code","source":"train1 = train[:80000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# look for users \n\ntrainmerge[\"user_id\"][7000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# merge train and qns\ntrainmerge = train1.merge(questions_df, left_on='content_id', right_on='question_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# merge with lectures\ntrainmerge2 = trainmerge.merge(lectures_df, left_on='content_id', right_on='lecture_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainmerge2.head()\n# content type id False/0 is qn, True/1 is lecture","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user= trainmerge[trainmerge['user_id']==1232090]\nlen(user)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# find match\nuser_q = user[user['answered_correctly']!=-1]\nlen(user_q)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_q.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#rowids\ntemp_rowids[1:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_q = user # include lectures \n\n# array to hold probabilities\nprob = []\nprobcount= 0\n\n#user_q.head()\n# P_forgetting\n# for each incorrect answer, match with tags that came before it\n\n# compare a forgotten entry with non-forgotten entries\n\nincorrect_row = user_q[user_q['answered_correctly']==0]\n#incorrect_row\nrowids = incorrect_row['row_id']\ntemp_rowids=rowids\n\ntags_rows = []\n\n#for r in incorrect_row:\n    #current_row_id = r[0]\n    #print(r)\n    # share at least 2 tags \n    \n    # share at least 3 tags\n\nctot = 0\nctotcorr =0\n\nfor ind in user_q.index:\n    ctot= ctot+1\n    # check correctness\n    curr_correct = user_q['answered_correctly'][ind]\n    if curr_correct == 1:\n        ctotcorr = ctotcorr+1\n    \n    irow = user_q['row_id'][ind]\n    ts = user_q['tags'][ind]\n    tsplit = ts.split(' ')\n    \n    temptags = []\n    for t in tsplit:\n        temptags.append(int(t))\n    \n    #print(tsplit)\n    #print(temptags)\n    \n    #tags_rows.append(temptags) #tsplit\n    tempprobcount = 0\n    if irow== int(temp_rowids[0:1]):\n        # compare\n        \n        for tarray in tags_rows:\n            tcount =0\n            for tg in tarray:\n                if tg in temptags:\n                    tcount=tcount+1\n            if tcount >= 2:\n                print(\"tags more than 2 found\")\n                tempprobcount=tempprobcount+1\n        # add prob to array\n        #pval = float(tempprobcount/ctot)\n        pval = float(tempprobcount/ctotcorr)\n        prob.append(pval)\n        \n        #remove\n        temp_rowids =temp_rowids[1:]\n        continue\n    \n    # append\n    tags_rows.append(temptags)\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nprob\n# filter by nonzero values?\nnp.mean(prob)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"p_slip : algorithm for p_forget, but also look at lectures to see if tag is there"},{"metadata":{},"cell_type":"markdown","source":"p_learned: \nlook for correct answers with new (unique) tags"},{"metadata":{"trusted":true},"cell_type":"code","source":"# look for unique tags\n\ncorrect_row = user_q[user_q['answered_correctly']==1]\ncorrect_rowids = correct_row['row_id']\n\ntags_rows = []\n# array to hold all the tags previously found \nalltags = []\n\nnumunique=0\n\nnrows = 0\nfor ind in user_q.index:\n    nrows=nrows+1\n    irow = user_q['row_id'][ind]\n    ts = user_q['tags'][ind]\n    # correct?\n    answercorrect = user_q['answered_correctly'][ind]\n    \n    tsplit = ts.split(' ')\n    \n    uniquebool = 1\n    #if irow== int(correct_rowids[0:1]):\n    if answercorrect==1:\n        for t in tsplit:\n            if t in alltags:\n                #found not unique\n                uniquebool=0\n                #continue\n        # if unique\n        if uniquebool==1:\n            print(\"unique\")\n            numunique = numunique+1\n            continue\n    \n    if uniquebool==0:\n        continue\n    \n    for t in tsplit:\n        alltags.append(int(t))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_q[3:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"nrows","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"numunique","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user = user[user['answered_correctly']!=-1]\n#len(user1512827)\n#user1512827 = user1512827[:70]\ncorrectly = user['answered_correctly']\nlen(user)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weightcorrect= np.multiply(correctly, wtarray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"correct128 = user128['answered_correctly']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weightcorrect128 = np.multiply(correct128, wtarray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior= user['prior_question_elapsed_time']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weightcorrect5382 = np.multiply(correctly, wtarray)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"wtarray= []\nfor t in user['tags']: # user138\n    #print(t)\n    ts = t.split(' ')\n    #print(ts[0])\n    parray = []\n    for j in ts:\n        nj = int(j)\n        tv = tags_df[tags_df['tag']==nj]\n        #print(tv)\n        v = tv['Percent_correct'] # change to 1 - percent\n        dv = tv['diffweight']\n        #print(float(v))\n        parray.append(float(dv))\n    #ave\n    #print(\"ave is\", np.mean(parray))\n    wtarray.append(np.mean(parray))\n        ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"save output\nhttps://www.kaggle.com/getting-started/60617"},{"metadata":{"trusted":true},"cell_type":"code","source":"# transform correct score yt1 by multiplying by the difficulty of the question answered\n# average difficulty of total question tags in the row \n\n# within a seq of lectures how user performs ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xt = np.arange(30)\nimport matplotlib.pyplot as plt\n#plt.scatter(xt, yt)\nplt.scatter(xt, weightcorrect)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior.to_csv('time123.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weightcorrect.to_csv('wt123.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a href=\"time123.csv\"> Download File </a>"},{"metadata":{"trusted":true},"cell_type":"code","source":"xt = np.arange(13)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n#plt.scatter(xt, yt)\nplt.scatter(xt, weightcorrect138)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from statsmodels.graphics.tsaplots import plot_acf\nplot_acf(weightcorrect138, lags=5)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# remove the lectures \ntrain = train[train['answered_correctly']!= -1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# subset of train\ntrain = train[:100000]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"save data for anova analysis "},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[\"content_type_id\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import seaborn as sns\ng=sns.countplot(train.content_type_id, palette='autumn')\n\n# False = 0 , True = 1 ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot the time that user spends on app\n\n# pick a user\nuser115 = train[train['user_id']==115]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user117 = train[train['user_id']==117]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#user115['timestamp']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\ntimedelta = pd.to_timedelta(train['timestamp'], unit='ms')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[train['row_id']==46]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"timedelta[46]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"timedelta.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"https://www.kaggle.com/datafan07/riiid-challenge-eda-baseline-model\ntaken from\n\nusers and content\n\nHere we have unique ID for each user , if we count all the interections made by user we can see there are some pretty active users, out of ~300k unique users we see top 25 of them almost made more than 3k interactions\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfig, ax = plt.subplots(ncols=1, nrows=1, figsize=(32,14))\n\n# distplot\n\nsns.countplot(y='user_id', data=train, order=train.user_id.value_counts().index[:25], palette='autumn',ax = ax[0])\nax[0].set_title('Top 25 Active Users', weight='bold')\n\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(ncols=2, nrows=1, figsize=(32,14))\n\n# distplot\n\nsns.countplot(y='user_id', data=train, order=train.user_id.value_counts().index[:25], palette='autumn',ax = ax[0])\nax[0].set_title('Top 25 Active Users', weight='bold')\n\n# countplot\n\nsns.countplot(y='content_id', data=train, order=train.content_id.value_counts().index[:25], palette='rocket',ax = ax[1])\nax[1].set_title('Top 25 Content', weight='bold')\n\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"task containers\n\n\"task_container_id\": Id code for the batch of questions or lectures. For example, a user might see three questions in a row before seeing the explanations for any of them. Those three would all share a task_container_id.\n    \nTasks with smaller id's are more common than bigger #s"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(ncols=2, nrows=1, figsize=(32,14))\n\n# distribution\n\nsns.distplot(train.task_container_id, kde=False,hist_kws={\n                 'rwidth': 0.85,\n                 'edgecolor': 'black',\n                 'alpha': 0.8}, ax=ax[0])\n\n\nax[0].set_ylabel('Frequency')\nax[0].set_title('Task Container ID Distribution', weight='bold')\n\n# counts\n\nsns.countplot(y='task_container_id', data=train_df, order=train_df.task_container_id.value_counts().index[:25], palette='cubehelix', ax=ax[1])\nax[1].set_title('Top 25 Tasks', weight='bold')\n\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"top 25 active users\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\n#grouping by user id and getting mean, sum, counts\n\nusr_ans = train.groupby('user_id').agg({ 'answered_correctly': ['mean','sum', 'count']})\nusr_ans.columns = ['avg_correct_answer','num_of_correct', 'total_answers']\n\n# changing dtype for reducing memory (default = 64)\n\nusr_ans['num_of_correct'] = usr_ans['num_of_correct'].astype('int16')\nusr_ans['total_answers'] = usr_ans['total_answers'].astype('int16')\n\n\ntrain_df = pd.merge(train, usr_ans, how='left', on = 'user_id')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Record the accuracy of the user \n"},{"metadata":{"trusted":true},"cell_type":"code","source":"ans_list = usr_ans[usr_ans['total_answers']>10].sort_values('avg_correct_answer', ascending=False).reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# look up avg correct answer by user id \nans_list.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ans_list[250:253]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user= train[train['user_id']==422628]\nlen(user)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior = user['prior_question_elapsed_time']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior.to_csv('time40to50.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a href=\"time40to50.csv\"> Download File </a>"},{"metadata":{"trusted":true},"cell_type":"code","source":"ans_list.loc[ans_list['user_id']==1167134087]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# go through ans_list\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# linear reg with one predictor: user, y: probability it is correct\n\n# predictor: time spent on the app\n\n#https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LinearRegression.html\n\nimport numpy as np\nfrom sklearn.linear_model import LinearRegression\n\nreg = LinearRegression().fit(X, y)\n\n\nreg.predict(np.array([[3, 5]]))\n\n# predictor: number of qns answered \n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"https://www.kaggle.com/erikbruin/riiid-eda-of-full-dataset\n I can add up all Wrong and Right answers for all questions that are labeled with a particular tag and calculate the percent correct for each tag\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df[questions_df.tags.isna()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df['tags'] = questions_df['tags'].astype(str)\n\ntags = [x.split() for x in questions_df[questions_df.tags != \"nan\"].tags.values]\ntags = [item for elem in tags for item in elem]\ntags = set(tags)\ntags = list(tags)\nprint(f'There are {len(tags)} different tags')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# look at difficulty of question\n# separate tags dataframe\n\ntags_df = pd.DataFrame()\nfor x in range(len(tags)):\n    df = pd.DataFrame()\n    for y in range(len(questions_df)):\n        if (tags[x] in questions_df.tags.values[y]):\n            df = df.append(questions_df.iloc[y,:])\n\n    df1 = df.agg({'Wrong': ['sum'], 'Right': ['sum']})\n    df1['Total_questions'] = df1.Wrong + df1.Right\n    df1['Question_ids_with_tag'] = len(df)\n    df1['tag'] = tags[x]\n    df1 = df1.set_index('tag')\n    tags_df = tags_df.append(df1)\n\ntags_df[['Wrong', 'Right', 'Total_questions']] = tags_df[['Wrong', 'Right', 'Total_questions']].astype(int)\ntags_df['Percent_correct'] = tags_df.Right/tags_df.Total_questions\ntags_df = tags_df.sort_values(by = \"Percent_correct\")\n\ntags_df.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#  plotting top 25 Accurate Users\nimport matplotlib.pyplot as plt\nsns.barplot(x='avg_correct_answer',y='user_id', orient='h', data=usr_ans[usr_ans['total_answers']>100].sort_values('avg_correct_answer', ascending=False).reset_index().iloc[:25],\n            palette='rocket', order=usr_ans[usr_ans['total_answers']>100].sort_values('avg_correct_answer', ascending=False).reset_index().user_id.iloc[:25])\nplt.title('Top 25 Accurate Users', weight='bold')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Answer Accuracy vs. Total Questions Answered\n\nHere we can observe increasing correct answer ratio by total number of questions answered by the user. Practice makes perfect!"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.regplot(data=usr_ans[usr_ans['total_answers']> 100], y='avg_correct_answer', x='total_answers', ci=False, scatter_kws={'alpha':0.5}, line_kws={\"color\": \"orange\"})\nplt.axhline(train_df.avg_correct_answer.mean(), color='k', linestyle='dashed', linewidth=3)\nplt.axvline(train_df.total_answers.mean(), color='k', linestyle='dashed', linewidth=3)\n\nmin_ylim, max_ylim = plt.ylim()\nplt.text(train_df.total_answers.mean()+25, max_ylim*0.20, 'Average Questions Solved {:.2f}'.format(train_df.total_answers.mean()))\nplt.text(train_df.total_answers.mean()+2400, max_ylim*0.6, 'Average Correct Answer: {:.2f}'.format(train_df.avg_correct_answer.mean()))\n\nplt.title('Average Correct Answer Ratio vs. Total Questions Answered per User', weight='bold')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Baseline\n- start by filling some missing values in our data and then split it as X and y for modelling. \n- predict if the given user answered specific question correct or failed it.\n"},{"metadata":{},"cell_type":"markdown","source":"X = train.copy()\n\n# fill na"},{"metadata":{"trusted":true},"cell_type":"code","source":"train['prior_question_elapsed_time'].fillna(0,  inplace=True)\ntrain['prior_question_had_explanation'] = train['prior_question_had_explanation'].fillna(value = False).astype(bool)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=train.sort_values(['user_id'])\ny = X[[\"answered_correctly\"]]\nX = X.drop([\"answered_correctly\"], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n#(test_df, sample_prediction_df) = next(iter_test)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# getting numberical labels for categorical data\n\nlb_make = LabelEncoder()\nX[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(X[\"prior_question_had_explanation\"])\nX['prior_question_had_explanation_enc'] = X['prior_question_had_explanation_enc'].astype('int8')\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# select features for training\n\nX = X[['avg_correct_answer','num_of_correct','total_answers', 'avg_correct_answer_c', 'prior_question_elapsed_time','prior_question_had_explanation_enc','part']] ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load models\nfrom sklearn.model_selection import StratifiedKFold, cross_validate\nimport lightgbm as lgb\n\nlight = lgb.LGBMClassifier()\n\nfrom sklearn.linear_model import LogisticRegression\n\n# import models\n\n#naive bayes\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"https://stats.stackexchange.com/questions/89914/building-a-classification-model-for-strictly-binary-data\nbinary data classification\n"},{"metadata":{},"cell_type":"markdown","source":"Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# fit model\nlight.fit(X, y)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"light.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"light.score(X_test, y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import riiideducation\n\nenv = riiideducation.make_env()\n\niter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lightpred = light.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"test out other classifiers\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"# naive bayes\nfrom sklearn.naive_bayes import GaussianNB\ngnb = GaussianNB()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gnb.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gnb.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gnb.score(X_test, y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlog_clf = LogisticRegression(random_state=0).fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#clf.predict(X_test)\nlog_clf.score(X_test, y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"logpred = log_clf.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# score on test data set\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#dec tree\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import cross_val_score\n\ndt_clf = DecisionTreeClassifier(random_state=0)\n#cross_val_score(clf, iris.data, iris.target, cv=10)\n\ndt_clf.fit(X_train, y_train)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#dt_clf.predict(X_test)\ndt_clf.score(X_test, y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# support vec machine\nfrom sklearn import svm\nclfsvm = svm.SVC()\nclfsvm.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clfsvm.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clfsvm.score(X_test, y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier\nclfada = AdaBoostClassifier(n_estimators=100, random_state=0)\nclfada.fit(X_train, y_train)\n\nclfada.score(X_test, y_test)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"adapred = clfada.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    print(y_pred)\n    y_pred = light.predict_proba(test_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Group prediction by content_type_id is 0: the event was a question  (otherwise 1 if lecture)\n"},{"metadata":{},"cell_type":"markdown","source":"hardest and easiest tags\nIf difficulty is greater , less probability to answer correctly\n\n[https://www.kaggle.com/erikbruin/riiid-eda-of-full-dataset](http://)"},{"metadata":{"trusted":true},"cell_type":"code","source":"select_rows = list(range(0,10)) + list(range(178, tags_df.shape[0]))\ntags_select = tags_df.iloc[select_rows,2]\n\nfig = plt.figure(figsize=(12,6))\nx = tags_select.index\ny = tags_select.values\nclrs = ['red' if y < 0.6 else 'green' for y in tags_select.values]\ntags_select.plot.bar(x, y, color=clrs)\nplt.title(\"Ten hardest and ten easiest tags\")\nplt.xlabel(\"Tag\")\nplt.ylabel(\"Percent answers correct of questions with the tag\")\nplt.xticks(rotation=90)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"score:"},{"metadata":{},"cell_type":"markdown","source":"from sklearn.metrics import roc_auc_score\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc_auc_score(train, lgbm.predict_proba(train_part_df[features])[:,1])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Extract time series of user's score(?) through time "},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, accuracy_score\n\n\n#y_pred = classifier.predict(X_test)\n\n#acc =  accuracy_score(y_test, y_pred)\n\ncm = confusion_matrix(y_test, lightpred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import plot_confusion_matrix\nimport matplotlib.pyplot as plt\nplot_confusion_matrix(light, X_test, y_test, cmap=plt.cm.Purples) # cmap=plt.cm.Blues","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pip install scikit-plot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import scikitplot as skplt\nskplt.metrics.plot_confusion_matrix(y_test, logpred,normalize=True, cmap=plt.cm.YlGnBu)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# color maps https://matplotlib.org/tutorials/colors/colormaps.html\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}