{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Section 1. Feature Engineering"},{"metadata":{},"cell_type":"markdown","source":"## 1.1 Data Reading"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-input":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2020-11-21T09:28:35.604145Z","iopub.status.busy":"2020-11-21T09:28:35.60329Z","iopub.status.idle":"2020-11-21T09:28:36.601674Z","shell.execute_reply":"2020-11-21T09:28:36.602413Z"},"papermill":{"duration":1.087477,"end_time":"2020-11-21T09:28:36.602606","exception":false,"start_time":"2020-11-21T09:28:35.515129","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.style as style\nstyle.use('fivethirtyeight') # https://matplotlib.org/3.1.0/gallery/style_sheets/style_sheets_reference.html\nimport seaborn as sns\nimport os\nfrom matplotlib.ticker import FuncFormatter\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"At the begining, we spent lots of time of reading the csv files, but it crashed on all kinds on devices, no matter personal server or google colab. Hence, we reference to online kaggle notebook for data reading and also some guidelines for basic feature engineering"},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:28:36.913825Z","iopub.status.busy":"2020-11-21T09:28:36.913061Z","iopub.status.idle":"2020-11-21T09:28:53.539135Z","shell.execute_reply":"2020-11-21T09:28:53.538375Z"},"papermill":{"duration":16.711,"end_time":"2020-11-21T09:28:53.539258","exception":false,"start_time":"2020-11-21T09:28:36.828258","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n\ntrain = pd.read_pickle(\"../input/riiid-train-data-multiple-formats/riiid_train.pkl.gzip\")\n\nprint(\"Train size:\", train.shape)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:29:21.842735Z","iopub.status.busy":"2020-11-21T09:29:21.841079Z","iopub.status.idle":"2020-11-21T09:29:22.246739Z","shell.execute_reply":"2020-11-21T09:29:22.246123Z"},"papermill":{"duration":4.685343,"end_time":"2020-11-21T09:29:22.246871","exception":false,"start_time":"2020-11-21T09:29:17.561528","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# Accroding to the original notebook's author, this saved lots of memory just by transforming the data type of a column\ntrain['prior_question_had_explanation'] = train['prior_question_had_explanation'].astype('bool')","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:22.567518Z","iopub.status.busy":"2020-11-21T09:29:22.566736Z","iopub.status.idle":"2020-11-21T09:29:22.597034Z","shell.execute_reply":"2020-11-21T09:29:22.596225Z"},"papermill":{"duration":0.116174,"end_time":"2020-11-21T09:29:22.597198","exception":false,"start_time":"2020-11-21T09:29:22.481024","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nexample_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\nexample_sample_submission = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 1.2 Exploratory Data Analysis\nFeatures:\n- Index Column: **row_id**, **user_id**\n- Useful column: **timestamp**, **content_id**, **content_type_id**, **task_container_id**, **user_answer**, **prior_question_elapsed_time**, **prior_question_had_explination**\n\nFrom here, we started to explore each column with basic statistics and plotting."},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:29:22.932337Z","iopub.status.busy":"2020-11-21T09:29:22.931213Z","iopub.status.idle":"2020-11-21T09:29:22.946337Z","shell.execute_reply":"2020-11-21T09:29:22.945621Z"},"papermill":{"duration":0.113932,"end_time":"2020-11-21T09:29:22.946465","exception":false,"start_time":"2020-11-21T09:29:22.832533","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:23.125964Z","iopub.status.busy":"2020-11-21T09:29:23.124427Z","iopub.status.idle":"2020-11-21T09:29:23.921085Z","shell.execute_reply":"2020-11-21T09:29:23.920081Z"},"papermill":{"duration":0.881805,"end_time":"2020-11-21T09:29:23.921253","exception":false,"start_time":"2020-11-21T09:29:23.039448","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('In train set, we have')\nprint('    {} unique users'.format(train.user_id.nunique()))\nprint('    {} contents'.format(train.content_id.nunique()))\nprint('    {} task containers'.format(train.task_container_id.nunique()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### content_type_id\n- (int8) 0 if the event was a question being posed to the user, 1 if the event was the user watching a lecture."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:24.247317Z","iopub.status.busy":"2020-11-21T09:29:24.246511Z","iopub.status.idle":"2020-11-21T09:29:25.43109Z","shell.execute_reply":"2020-11-21T09:29:25.430352Z"},"papermill":{"duration":1.26857,"end_time":"2020-11-21T09:29:25.431212","exception":false,"start_time":"2020-11-21T09:29:24.162642","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# True - a question  /  False - a lecture\ncontent_type_stat = train.content_type_id.value_counts()\ncontent_type_stat.index = ['lecture', 'question']\nprint(content_type_stat)\ncontent_type_stat.plot.bar()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We can find that in all contents, questions only has no more than 1 percent."},{"metadata":{"papermill":{"duration":0.079925,"end_time":"2020-11-21T09:29:25.59158","exception":false,"start_time":"2020-11-21T09:29:25.511655","status":"completed"},"tags":[]},"cell_type":"markdown","source":"#### Content_id \n- (int16) ID code for the user interaction\n\nIf content_type is question, this column also refers to question."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:25.758854Z","iopub.status.busy":"2020-11-21T09:29:25.757344Z","iopub.status.idle":"2020-11-21T09:29:34.081083Z","shell.execute_reply":"2020-11-21T09:29:34.080425Z"},"papermill":{"duration":8.409426,"end_time":"2020-11-21T09:29:34.081208","exception":false,"start_time":"2020-11-21T09:29:25.671782","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print(f'We have {train.content_id.nunique()} content ids in our train set, of which {train[train.content_type_id == False].content_id.nunique()} are questions.')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"By observing the top 50, we found 2 interesting facts.\n1. there are several \"drop\" or user count at certain n. For example, a relatively huge drop between top 26 & top 27\n2. for different interval, the drop's \"slope\" differ a lot. For example, top 1 ~ top 6 has different drop interval compares to top 7 ~ top 20. Some of them are sharp and some of them are flat."},{"metadata":{"trusted":true},"cell_type":"code","source":"top_50_content = train.content_id.value_counts()[:50]\ntop_50_content.index = [str(ele) for ele in list(range(1, 51))]\ntop_50_content.plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.081707,"end_time":"2020-11-21T09:29:38.657356","exception":false,"start_time":"2020-11-21T09:29:38.575649","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### task_container_id \n- (int16) Id code for the batch of questions or lectures. For example, a user might see three questions in a row before seeing the explanations for any of them. Those three would all share a task_container_id."},{"metadata":{"trusted":true},"cell_type":"code","source":"top_50_task = train.task_container_id.value_counts()[:50]\ntop_50_task.index = [str(ele) for ele in list(range(1, 51))]\ntop_50_task.plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.08345,"end_time":"2020-11-21T09:29:39.868329","exception":false,"start_time":"2020-11-21T09:29:39.784879","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### User answer\n- (int8) the user's answer to the question, if any. Read -1 as null, for lectures.\n\nInsight: from figure below, we find that there are 5 unique values for **user_answer**, and \"-1\" stands for lecture. Other answers distributed evenly from 0 ~ 3."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:40.041287Z","iopub.status.busy":"2020-11-21T09:29:40.040499Z","iopub.status.idle":"2020-11-21T09:29:41.2259Z","shell.execute_reply":"2020-11-21T09:29:41.22511Z"},"papermill":{"duration":1.27386,"end_time":"2020-11-21T09:29:41.226028","exception":false,"start_time":"2020-11-21T09:29:39.952168","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_ans_stat = train.user_answer.value_counts()\nuser_ans_stat.plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.083062,"end_time":"2020-11-21T09:29:41.394104","exception":false,"start_time":"2020-11-21T09:29:41.311042","status":"completed"},"tags":[]},"cell_type":"markdown","source":"### timestamp\n- (int64) the time in milliseconds between this user interaction and the first event completion from that user.\n\nInsight: From the author's code, we find that most user "},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:41.587596Z","iopub.status.busy":"2020-11-21T09:29:41.586325Z","iopub.status.idle":"2020-11-21T09:29:55.833Z","shell.execute_reply":"2020-11-21T09:29:55.832269Z"},"papermill":{"duration":14.355982,"end_time":"2020-11-21T09:29:55.833134","exception":false,"start_time":"2020-11-21T09:29:41.477152","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"#1 year = 31536000000 ms\nts = train['timestamp']/(31536000000/12)\nfig = plt.figure(figsize=(12,6))\nts.plot.hist(bins=100)\nplt.title(\"Histogram of timestamp\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Months between this user interaction and the first event completion from that user\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.085513,"end_time":"2020-11-21T09:29:57.438202","exception":false,"start_time":"2020-11-21T09:29:57.352689","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# The target: answered_correctly\n\nOur target focusing on correctly predicting the column **answered_correctly**. We look from different aspects(generally, task_container_id, user_id) into this column in order to find some clues for feature enginerring."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:29:57.618538Z","iopub.status.busy":"2020-11-21T09:29:57.617207Z","iopub.status.idle":"2020-11-21T09:30:03.692644Z","shell.execute_reply":"2020-11-21T09:30:03.691913Z"},"papermill":{"duration":6.168462,"end_time":"2020-11-21T09:30:03.69278","exception":false,"start_time":"2020-11-21T09:29:57.524318","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"answer_stat = train[train.answered_correctly != -1].answered_correctly.value_counts()\nanswer_stat.index = ['correct', 'wrong']\n\nanswer_stat.plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"task_accuracy_stat = train.groupby('task_container_id')['answered_correctly'].agg(\n    average_accuracy = 'mean', \n    question_count = 'count'\n).reset_index()\n\ntask_accuracy_stat[['average_accuracy', 'question_count']].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(12,6))\n\ntask_accuracy_stat.average_accuracy.plot.hist(bins=50)\nplt.title(\"Histogram of average accuracy grouped by task container\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_accuracy_stat = train.groupby('user_id')['answered_correctly'].agg(\n    average_accuracy = 'mean', \n    question_count = 'count'\n).reset_index()\n\nuser_accuracy_stat[['average_accuracy', 'question_count']].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(12,6))\n\nuser_accuracy_stat.average_accuracy.plot.hist(bins=50)\nplt.title(\"Histogram of average accuracy grouped by user\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(12,6))\n\nuser_accuracy_stat.question_count.plot.hist(bins=50)\nplt.title(\"Histogram of question_count grouped by user\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.092386,"end_time":"2020-11-21T09:31:00.77193","exception":false,"start_time":"2020-11-21T09:31:00.679544","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## prior_question_had_explanation\n\n- (bool) Whether or not the user saw an explanation and the correct response(s) after answering the previous question bundle, ignoring any lectures in between. The value is shared across a single question bundle, and is null for a user's first question bundle or lecture. Typically the first several questions a user sees were part of an onboarding diagnostic test where they did not get any feedback.\n\nInsight: Intuitively, it's reasonable that explanation can help us on answering questions. Hence, we group by **prior_question_had_explanation** and calculate average **answered_correctly**. The outcome shows that with prior explanation, the average accuracy is 17% more than those question without prior explanation."},{"metadata":{"trusted":true},"cell_type":"code","source":"question_df = train[train.answered_correctly != -1]\nprior_exp_stat = question_df.groupby('prior_question_had_explanation')['answered_correctly'].agg(\n    average_accuracy = 'mean',\n)\nprior_exp_stat","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.094227,"end_time":"2020-11-21T09:31:24.722637","exception":false,"start_time":"2020-11-21T09:31:24.62841","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## prior_question_elapsed_time\n\n- (float32) The average time in milliseconds it took a user to answer each question in the previous question bundle, ignoring any lectures in between. Is null for a user's first question bundle or lecture. Note that the time is the average time a user took to solve each question in the previous bundle.\n\nInsight: from the statistics, we didn't see too much change betwwen the average and the standard deviation also didn't differ too much. There's no useful information currently from this column. Maybe the model can find some details when combining with other columns."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:31:24.92009Z","iopub.status.busy":"2020-11-21T09:31:24.918568Z","iopub.status.idle":"2020-11-21T09:31:43.10865Z","shell.execute_reply":"2020-11-21T09:31:43.108048Z"},"papermill":{"duration":18.291692,"end_time":"2020-11-21T09:31:43.108799","exception":false,"start_time":"2020-11-21T09:31:24.817107","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"from scipy import stats\n\ntime_df = question_df[['answered_correctly', 'prior_question_elapsed_time']]\n\ntime_df.groupby('answered_correctly')['prior_question_elapsed_time'].agg(\n    average_time = 'mean',\n    std_time = 'std'\n)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.094656,"end_time":"2020-11-21T09:31:43.299988","exception":false,"start_time":"2020-11-21T09:31:43.205332","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# 1.2 Exploring Questions\n\nMetadata for the questions posed to users.\n\n* question_id: foreign key for the train/test content_id column, when the content type is question (0).\n* bundle_id: code for which questions are served together.\n* correct_answer: the answer to the question. Can be compared with the train user_answer column to check if the user was right.\n* part: the relevant section of the TOEIC test.\n* tags: one or more detailed tag codes for the question. The meaning of the tags will not be provided, but these codes are sufficient for clustering the questions together. 1 row has \"NA\" for tags\n\nKey: LEFT JOIN train.content_id ON questions.question_id"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:31:43.504679Z","iopub.status.busy":"2020-11-21T09:31:43.503565Z","iopub.status.idle":"2020-11-21T09:31:43.507523Z","shell.execute_reply":"2020-11-21T09:31:43.508109Z"},"papermill":{"duration":0.111639,"end_time":"2020-11-21T09:31:43.50826","exception":false,"start_time":"2020-11-21T09:31:43.396621","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:31:43.706667Z","iopub.status.busy":"2020-11-21T09:31:43.705527Z","iopub.status.idle":"2020-11-21T09:31:43.710317Z","shell.execute_reply":"2020-11-21T09:31:43.709646Z"},"papermill":{"duration":0.106158,"end_time":"2020-11-21T09:31:43.710449","exception":false,"start_time":"2020-11-21T09:31:43.604291","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions.shape","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:31:45.905506Z","iopub.status.busy":"2020-11-21T09:31:45.904699Z","iopub.status.idle":"2020-11-21T09:31:45.928607Z","shell.execute_reply":"2020-11-21T09:31:45.927631Z"},"papermill":{"duration":0.137004,"end_time":"2020-11-21T09:31:45.928811","exception":false,"start_time":"2020-11-21T09:31:45.791807","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions['tags'] = questions['tags'].astype(str)\n\ntags = [x.split() for x in questions[questions.tags != \"nan\"].tags.values]\ntags = [item for elem in tags for item in elem]\ntags = set(tags)\ntags = list(tags)\nprint(f'There are {len(tags)} different tags')","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.103216,"end_time":"2020-11-21T09:31:46.153082","exception":false,"start_time":"2020-11-21T09:31:46.049866","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Let's find out how many answers were Right and Wrong per question_id (so per content_id in train)."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:31:46.483488Z","iopub.status.busy":"2020-11-21T09:31:46.481841Z","iopub.status.idle":"2020-11-21T09:31:59.254166Z","shell.execute_reply":"2020-11-21T09:31:59.253433Z"},"papermill":{"duration":13.002157,"end_time":"2020-11-21T09:31:59.254295","exception":false,"start_time":"2020-11-21T09:31:46.252138","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"tags_list = [x.split() for x in questions.tags.values]\nquestions['tags'] = tags_list\nquestions.head()\n\ncorrect = train[train.answered_correctly != -1].groupby([\"content_id\", 'answered_correctly'], as_index=False).size()\ncorrect = correct.pivot(index= \"content_id\", columns='answered_correctly', values='size')\ncorrect.columns = ['Wrong', 'Right']\ncorrect = correct.fillna(0)\ncorrect[['Wrong', 'Right']] = correct[['Wrong', 'Right']].astype(int)\nquestions = questions.merge(correct, left_on = \"question_id\", right_on = \"content_id\", how = \"left\")\nquestions.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.099816,"end_time":"2020-11-21T09:31:59.455444","exception":false,"start_time":"2020-11-21T09:31:59.355628","status":"completed"},"tags":[]},"cell_type":"markdown","source":"As you can see, I have also changed the tags column into lists of tags."},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:31:59.660163Z","iopub.status.busy":"2020-11-21T09:31:59.659318Z","iopub.status.idle":"2020-11-21T09:31:59.664382Z","shell.execute_reply":"2020-11-21T09:31:59.663661Z"},"papermill":{"duration":0.109843,"end_time":"2020-11-21T09:31:59.664521","exception":false,"start_time":"2020-11-21T09:31:59.554678","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions.tags.values","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.099955,"end_time":"2020-11-21T09:31:59.86415","exception":false,"start_time":"2020-11-21T09:31:59.764195","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Now, I can add up all Wrong and Right answers for all questions that are labeled with a particular tag and calculate the percent correct for each tag. Please note that there is \"double counting\" of questions; for instance if a question has 5 tags, its answers are aggregated in the totals of each of the 5 tags. "},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:32:00.111411Z","iopub.status.busy":"2020-11-21T09:32:00.106629Z","iopub.status.idle":"2020-11-21T09:34:49.76241Z","shell.execute_reply":"2020-11-21T09:34:49.761308Z"},"papermill":{"duration":169.797697,"end_time":"2020-11-21T09:34:49.762677","exception":false,"start_time":"2020-11-21T09:31:59.96498","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n\ntags_df = pd.DataFrame()\nfor x in range(len(tags)):\n    df = pd.DataFrame()\n    for y in range(len(questions)):\n        if (tags[x] in questions.tags.values[y]):\n            df = df.append(questions.iloc[y,:])\n\n    df1 = df.agg({'Wrong': ['sum'], 'Right': ['sum']})\n    df1['Total_questions'] = df1.Wrong + df1.Right\n    df1['Question_ids_with_tag'] = len(df)\n    df1['tag'] = tags[x]\n    df1 = df1.set_index('tag')\n    tags_df = tags_df.append(df1)\n\ntags_df[['Wrong', 'Right', 'Total_questions']] = tags_df[['Wrong', 'Right', 'Total_questions']].astype(int)\ntags_df['Percent_correct'] = tags_df.Right/tags_df.Total_questions\ntags_df = tags_df.sort_values(by = \"Percent_correct\")\n\ntags_df.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.102355,"end_time":"2020-11-21T09:34:49.968032","exception":false,"start_time":"2020-11-21T09:34:49.865677","status":"completed"},"tags":[]},"cell_type":"markdown","source":"As you can see, the differences are significant!"},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:34:50.185324Z","iopub.status.busy":"2020-11-21T09:34:50.182935Z","iopub.status.idle":"2020-11-21T09:34:50.450137Z","shell.execute_reply":"2020-11-21T09:34:50.450686Z"},"papermill":{"duration":0.380193,"end_time":"2020-11-21T09:34:50.450889","exception":false,"start_time":"2020-11-21T09:34:50.070696","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"select_rows = list(range(0,10)) + list(range(178, len(tags_df)))\ntags_select = tags_df.iloc[select_rows,4]\n\nfig = plt.figure(figsize=(12,6))\nx = tags_select.index\ny = tags_select.values\nclrs = ['red' if y < 0.6 else 'green' for y in tags_select.values]\ntags_select.plot.bar(x, y, color=clrs)\nplt.title(\"Ten hardest and ten easiest tags\")\nplt.xlabel(\"Tag\")\nplt.ylabel(\"Percent answers correct of questions with the tag\")\nplt.xticks(rotation=90)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.103862,"end_time":"2020-11-21T09:34:50.658951","exception":false,"start_time":"2020-11-21T09:34:50.555089","status":"completed"},"tags":[]},"cell_type":"markdown","source":"However, we should also realize that the tag with the worst percent_correct only has about 250,000 answers. This a low number compared to the tags with most answers."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:34:50.878533Z","iopub.status.busy":"2020-11-21T09:34:50.87756Z","iopub.status.idle":"2020-11-21T09:34:51.203679Z","shell.execute_reply":"2020-11-21T09:34:51.203081Z"},"papermill":{"duration":0.440889,"end_time":"2020-11-21T09:34:51.203832","exception":false,"start_time":"2020-11-21T09:34:50.762943","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"tags_select = tags_df.sort_values(by = \"Total_questions\", ascending = False).iloc[:30,:]\ntags_select = tags_select[\"Total_questions\"]\n\nfig = plt.figure(figsize=(12,6))\nax = tags_select.plot.bar()\nplt.title(\"Thirty tags with most questions answered\")\nplt.xticks(rotation=90)\nplt.ticklabel_format(style='plain', axis='y')\nax.get_yaxis().set_major_formatter(FuncFormatter(lambda x, p: format(int(x), ','))) #add thousands separator\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.10582,"end_time":"2020-11-21T09:34:51.41495","exception":false,"start_time":"2020-11-21T09:34:51.30913","status":"completed"},"tags":[]},"cell_type":"markdown","source":"What are the so-called \"Parts\"? When following the link provided in the data description we find out that this relates to a test.\n\n> The TOEIC L&R uses an optically-scanned answer sheet. There are 200 questions to answer in two hours in Listening (approximately 45 minutes, 100 questions) and Reading (75 minutes, 100 questions). \n\nThe listening section consists of Part 1-4 (Listening Section (approx. 45 minutes, 100 questions)).\n\nThe reading section consists of Part 5-7 (Reading Section (75 minutes, 100 questions))."},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:34:51.633653Z","iopub.status.busy":"2020-11-21T09:34:51.632778Z","iopub.status.idle":"2020-11-21T09:34:51.63671Z","shell.execute_reply":"2020-11-21T09:34:51.63734Z"},"papermill":{"duration":0.117131,"end_time":"2020-11-21T09:34:51.6375","exception":false,"start_time":"2020-11-21T09:34:51.520369","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions.part.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.104797,"end_time":"2020-11-21T09:34:51.847556","exception":false,"start_time":"2020-11-21T09:34:51.742759","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Below, I am displaying the percent correct by part."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:34:52.070376Z","iopub.status.busy":"2020-11-21T09:34:52.069515Z","iopub.status.idle":"2020-11-21T09:34:52.267909Z","shell.execute_reply":"2020-11-21T09:34:52.268458Z"},"papermill":{"duration":0.314195,"end_time":"2020-11-21T09:34:52.268623","exception":false,"start_time":"2020-11-21T09:34:51.954428","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"part = questions.groupby('part').agg({'Wrong': ['sum'], 'Right': ['sum']})\npart['Percent_correct'] = part.Right/(part.Right + part.Wrong)\npart = part.iloc[:,2]\n\nfig = plt.figure(figsize=(12,6))\npart.plot.bar()\nplt.title(\"Percent_correct by part\")\nplt.xlabel(\"Part\")\nplt.ylabel(\"Percent answers correct\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.108932,"end_time":"2020-11-21T09:34:52.486344","exception":false,"start_time":"2020-11-21T09:34:52.377412","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# 1.3 Exploring Lectures\n\nMetadata for the lectures watched by users as they progress in their education.\n* lecture_id: foreign key for the train/test content_id column, when the content type is lecture (1).\n* part: top level category code for the lecture.\n* tag: one tag codes for the lecture. The meaning of the tags will not be provided, but these codes are sufficient for clustering the lectures together.\n* type_of: brief description of the core purpose of the lecture\n"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:34:52.713799Z","iopub.status.busy":"2020-11-21T09:34:52.712918Z","iopub.status.idle":"2020-11-21T09:34:52.71747Z","shell.execute_reply":"2020-11-21T09:34:52.716756Z"},"papermill":{"duration":0.122973,"end_time":"2020-11-21T09:34:52.717588","exception":false,"start_time":"2020-11-21T09:34:52.594615","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"lectures.head()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:34:52.947553Z","iopub.status.busy":"2020-11-21T09:34:52.946661Z","iopub.status.idle":"2020-11-21T09:34:52.951197Z","shell.execute_reply":"2020-11-21T09:34:52.950293Z"},"papermill":{"duration":0.124087,"end_time":"2020-11-21T09:34:52.951379","exception":false,"start_time":"2020-11-21T09:34:52.827292","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print(f'There are {lectures.shape[0]} lecture_ids.')","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.111723,"end_time":"2020-11-21T09:34:53.200462","exception":false,"start_time":"2020-11-21T09:34:53.088739","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Let's have a look at the type_of."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:34:53.44229Z","iopub.status.busy":"2020-11-21T09:34:53.437959Z","iopub.status.idle":"2020-11-21T09:34:53.59458Z","shell.execute_reply":"2020-11-21T09:34:53.593836Z"},"papermill":{"duration":0.283545,"end_time":"2020-11-21T09:34:53.5947","exception":false,"start_time":"2020-11-21T09:34:53.311155","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"lect_type_of = lectures.type_of.value_counts()\n\nfig = plt.figure(figsize=(12,6))\nlect_type_of .plot.bar()\nplt.title(\"Types of lectures\")\nplt.xlabel(\"type_of\")\nplt.ylabel(\"Count lecture_id\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.109712,"end_time":"2020-11-21T09:34:53.815072","exception":false,"start_time":"2020-11-21T09:34:53.70536","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Since there are not that many lectures, I want to check if it helps if a user watches lectures at all. As you can see, it helps indeed!"},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:34:54.049762Z","iopub.status.busy":"2020-11-21T09:34:54.048928Z","iopub.status.idle":"2020-11-21T09:35:01.210561Z","shell.execute_reply":"2020-11-21T09:35:01.21135Z"},"papermill":{"duration":7.286859,"end_time":"2020-11-21T09:35:01.21154","exception":false,"start_time":"2020-11-21T09:34:53.924681","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_lect = train.groupby([\"user_id\", \"answered_correctly\"]).size().unstack()\nuser_lect.columns = ['Lecture', 'Wrong', 'Right']\nuser_lect['Lecture'] = user_lect['Lecture'].fillna(0)\nuser_lect = user_lect.astype('Int64')\nuser_lect['Watches_lecture'] = np.where(user_lect.Lecture > 0, True, False)\n\nwatches_l = user_lect.groupby(\"Watches_lecture\").agg({'Wrong': ['sum'], 'Right': ['sum']})\nprint(user_lect.Watches_lecture.value_counts())\n\nwatches_l['Percent_correct'] = watches_l.Right/(watches_l.Right + watches_l.Wrong)\n#del user_lect\n\nwatches_l = watches_l.iloc[:,2]\n\nfig = plt.figure(figsize=(12,4))\nwatches_l.plot.barh()\nplt.title(\"User watches lectures: Percent_correct\")\nplt.xlabel(\"Percent correct\")\nplt.ylabel(\"User watched at least one lecture\")\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:35:01.441264Z","iopub.status.busy":"2020-11-21T09:35:01.440282Z","iopub.status.idle":"2020-11-21T09:35:01.444969Z","shell.execute_reply":"2020-11-21T09:35:01.44432Z"},"papermill":{"duration":0.121841,"end_time":"2020-11-21T09:35:01.445097","exception":false,"start_time":"2020-11-21T09:35:01.323256","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"65244627/(65244627+34026673)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.113149,"end_time":"2020-11-21T09:35:01.670667","exception":false,"start_time":"2020-11-21T09:35:01.557518","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Batches (task_container_id) may also contain lectures, and I want to find out if there are any batches with high numbers of lectures."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:35:01.905902Z","iopub.status.busy":"2020-11-21T09:35:01.905105Z","iopub.status.idle":"2020-11-21T09:35:09.718903Z","shell.execute_reply":"2020-11-21T09:35:09.719907Z"},"papermill":{"duration":7.936569,"end_time":"2020-11-21T09:35:09.720156","exception":false,"start_time":"2020-11-21T09:35:01.783587","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"batch_lect = train.groupby([\"task_container_id\", \"answered_correctly\"]).size().unstack()\nbatch_lect.columns = ['Lecture', 'Wrong', 'Right']\nbatch_lect['Lecture'] = batch_lect['Lecture'].fillna(0)\nbatch_lect = batch_lect.astype('Int64')\nbatch_lect['Percent_correct'] = batch_lect.Right/(batch_lect.Wrong + batch_lect.Right)\nbatch_lect['Percent_lecture'] = batch_lect.Lecture/(batch_lect.Lecture + batch_lect.Wrong + batch_lect.Right)\nbatch_lect = batch_lect.sort_values(by = \"Percent_lecture\", ascending = False)\n\nprint(f'The highest number of lectures watched within a single task_container_id is {batch_lect.Lecture.max()}.')","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.117211,"end_time":"2020-11-21T09:35:09.956548","exception":false,"start_time":"2020-11-21T09:35:09.839337","status":"completed"},"tags":[]},"cell_type":"markdown","source":"As you can see below (table sorted on descending Percent_lecture), the percent of lectures of the task_container_id's is never high. We can also see the highest percentages of lectures are around 2.8%, which means one lecture on about 36 questions."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:35:10.207939Z","iopub.status.busy":"2020-11-21T09:35:10.202433Z","iopub.status.idle":"2020-11-21T09:35:10.211903Z","shell.execute_reply":"2020-11-21T09:35:10.212524Z"},"papermill":{"duration":0.135888,"end_time":"2020-11-21T09:35:10.212677","exception":false,"start_time":"2020-11-21T09:35:10.076789","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"batch_lect.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.115568,"end_time":"2020-11-21T09:35:10.445074","exception":false,"start_time":"2020-11-21T09:35:10.329506","status":"completed"},"tags":[]},"cell_type":"markdown","source":"Is there a correlation between the percent_lecture and the percent_correct? No, I don't really see it. If anything, the percent_correct actually seems to go down slightly."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:35:10.693368Z","iopub.status.busy":"2020-11-21T09:35:10.692388Z","iopub.status.idle":"2020-11-21T09:35:10.909233Z","shell.execute_reply":"2020-11-21T09:35:10.909894Z"},"papermill":{"duration":0.348533,"end_time":"2020-11-21T09:35:10.910053","exception":false,"start_time":"2020-11-21T09:35:10.56152","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"batch = batch_lect.iloc[:, 3:]\n\nfig = plt.figure(figsize=(12,6))\nx = batch.Percent_lecture\ny = batch.Percent_correct\nplt.scatter(x, y, marker='o')\nplt.title(\"Percent lectures in a task_container versus percent answered correctly\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Percent lectures\")\nplt.ylabel(\"Percent answered correctly\")\n\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.118359,"end_time":"2020-11-21T09:35:11.152128","exception":false,"start_time":"2020-11-21T09:35:11.033769","status":"completed"},"tags":[]},"cell_type":"markdown","source":"The last thing that I want to check is if having a lecture in a batch helps. As you can see, it does not. Batches without lectures have about 8% more correct answers than batches with lectures."},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:35:11.397255Z","iopub.status.busy":"2020-11-21T09:35:11.396504Z","iopub.status.idle":"2020-11-21T09:35:11.403689Z","shell.execute_reply":"2020-11-21T09:35:11.402965Z"},"papermill":{"duration":0.132854,"end_time":"2020-11-21T09:35:11.403854","exception":false,"start_time":"2020-11-21T09:35:11.271","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"batch_lect['Has_lecture'] = np.where(batch_lect.Lecture == 0, False, True)\nprint(f'We have {batch_lect[batch_lect.Has_lecture == True].shape[0]} task_container_ids with lectures and {batch_lect[batch_lect.Has_lecture == False].shape[0]} task_container_ids without lectures.')","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-21T09:35:11.651911Z","iopub.status.busy":"2020-11-21T09:35:11.651108Z","iopub.status.idle":"2020-11-21T09:35:11.665429Z","shell.execute_reply":"2020-11-21T09:35:11.66628Z"},"papermill":{"duration":0.142775,"end_time":"2020-11-21T09:35:11.666441","exception":false,"start_time":"2020-11-21T09:35:11.523666","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"batch_lect = batch_lect[['Wrong', 'Right', 'Has_lecture']]\nbatch_lect = batch_lect.groupby(\"Has_lecture\").sum()\nbatch_lect['Percent_correct'] = batch_lect.Right/(batch_lect.Wrong + batch_lect.Right)\nbatch_lect = batch_lect[['Percent_correct']]\nbatch_lect","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Section 2. Modeling"},{"metadata":{"papermill":{"duration":0.122309,"end_time":"2020-11-21T09:35:15.297436","exception":false,"start_time":"2020-11-21T09:35:15.175127","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## 2.1 Baseline model"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:35:16.053471Z","iopub.status.busy":"2020-11-21T09:35:16.052508Z","iopub.status.idle":"2020-11-21T09:35:16.055596Z","shell.execute_reply":"2020-11-21T09:35:16.054878Z"},"papermill":{"duration":0.637722,"end_time":"2020-11-21T09:35:16.05575","exception":false,"start_time":"2020-11-21T09:35:15.418028","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"#this clears everything loaded in RAM, including the libraries\n%reset -f","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-input":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2020-11-21T09:35:16.312036Z","iopub.status.busy":"2020-11-21T09:35:16.311209Z","iopub.status.idle":"2020-11-21T09:35:16.484204Z","shell.execute_reply":"2020-11-21T09:35:16.483254Z"},"papermill":{"duration":0.308188,"end_time":"2020-11-21T09:35:16.484383","exception":false,"start_time":"2020-11-21T09:35:16.176195","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport riiideducation\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.style as style\nstyle.use('fivethirtyeight')\nimport seaborn as sns\nimport os\nimport lightgbm as lgb\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import LabelEncoder\nimport gc\nimport sys\npd.set_option('display.max_rows', None)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":false,"execution":{"iopub.execute_input":"2020-11-21T09:35:16.732833Z","iopub.status.busy":"2020-11-21T09:35:16.731869Z","iopub.status.idle":"2020-11-21T09:35:39.199339Z","shell.execute_reply":"2020-11-21T09:35:39.198647Z"},"papermill":{"duration":22.593906,"end_time":"2020-11-21T09:35:39.19947","exception":false,"start_time":"2020-11-21T09:35:16.605564","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\ncols_to_load = ['row_id', 'user_id', 'answered_correctly', 'content_id', 'prior_question_had_explanation', 'prior_question_elapsed_time']\ntrain = pd.read_pickle(\"../input/riiid-train-data-multiple-formats/riiid_train.pkl.gzip\")[cols_to_load]\ntrain['prior_question_had_explanation'] = train['prior_question_had_explanation'].astype('bool')\n\nprint(\"Train size:\", train.shape)\n\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nexample_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\nexample_sample_submission = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Total {} unique users with {} unique content'.format(train.user_id.nunique(), train.content_id.nunique()\n))\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Total {} unique questions'.format(questions.question_id.nunique()))\nprint('Total {} unique bundle'.format(questions.bundle_id.nunique()))\nquestions['tags_num'] = questions.tags.apply(lambda x: len(x.split(' ')) if isinstance(x, str) else 0)\nquestions.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions.tags_num.value_counts().plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Total {} unique lectures'.format(lectures.lecture_id.nunique()))\nprint(lectures.shape)\nlectures.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures.type_of.value_counts().plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:35:40.508398Z","iopub.status.busy":"2020-11-21T09:35:40.507525Z","iopub.status.idle":"2020-11-21T09:36:06.011939Z","shell.execute_reply":"2020-11-21T09:36:06.011181Z"},"papermill":{"duration":25.636487,"end_time":"2020-11-21T09:36:06.012066","exception":false,"start_time":"2020-11-21T09:35:40.375579","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n# adding user features\n# PK: user_id + user_questions\n# 1. remove lecture content, leaving only questions\n# 2. for each user, count how questions answered the average accuracy for them\n# 3. Join back train_df to add if user watches_lecture before\n\nuser_df = train[train.answered_correctly != -1].groupby('user_id').agg({'answered_correctly': ['count', 'mean']}).reset_index()\nuser_df.columns = ['user_id', 'user_questions', 'user_mean']\n\nuser_lect = train.groupby([\"user_id\", \"answered_correctly\"]).size().unstack()\nuser_lect.columns = ['Lecture', 'Wrong', 'Right']\nuser_lect['Lecture'] = user_lect['Lecture'].fillna(0)\nuser_lect = user_lect.astype('Int64')\nuser_lect['watches_lecture'] = np.where(user_lect.Lecture > 0, 1, 0)\nuser_lect = user_lect.reset_index()\nuser_lect = user_lect[['user_id', 'watches_lecture']]\n\nuser_df = user_df.merge(user_lect, on = \"user_id\", how = \"left\")\ndel user_lect\nuser_df.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:36:06.275427Z","iopub.status.busy":"2020-11-21T09:36:06.273952Z","iopub.status.idle":"2020-11-21T09:36:26.487216Z","shell.execute_reply":"2020-11-21T09:36:26.486575Z"},"papermill":{"duration":20.34853,"end_time":"2020-11-21T09:36:26.487338","exception":false,"start_time":"2020-11-21T09:36:06.138808","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n# adding content features\n# PK: content_id\n# 1. remove lecture content, leaving only questions\n# 2. count how many people answered the question & the average accuracy of each question\n\ncontent_df = train[train.answered_correctly != -1].groupby('content_id').agg({'answered_correctly': ['count', 'mean']}).reset_index()\ncontent_df.columns = ['content_id', 'content_questions', 'content_mean']\ncontent_df.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:36:26.999587Z","iopub.status.busy":"2020-11-21T09:36:26.998801Z","iopub.status.idle":"2020-11-21T09:36:42.513337Z","shell.execute_reply":"2020-11-21T09:36:42.512369Z"},"papermill":{"duration":15.648046,"end_time":"2020-11-21T09:36:42.513483","exception":false,"start_time":"2020-11-21T09:36:26.865437","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n#using one of the validation sets composed by tito\ncv2_train = pd.read_pickle(\"../input/riiid-cross-validation-files/cv2_train.pickle\")['row_id']\ncv2_valid = pd.read_pickle(\"../input/riiid-cross-validation-files/cv2_valid.pickle\")['row_id']","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:36:42.782215Z","iopub.status.busy":"2020-11-21T09:36:42.781328Z","iopub.status.idle":"2020-11-21T09:37:33.511328Z","shell.execute_reply":"2020-11-21T09:37:33.512028Z"},"papermill":{"duration":50.868204,"end_time":"2020-11-21T09:37:33.512206","exception":false,"start_time":"2020-11-21T09:36:42.644002","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train = train[train.answered_correctly != -1]\n\n#save mean before splitting\n#please be aware that there is an issues with train.prior_question_elapsed_time.mean()\n#see https://www.kaggle.com/c/riiid-test-answer-prediction/discussion/195032\nmean_prior = train.prior_question_elapsed_time.astype(\"float64\").mean()\n\nvalidation = train[train.row_id.isin(cv2_valid)]\ntrain = train[train.row_id.isin(cv2_train)]\n\nvalidation = validation.drop(columns = \"row_id\")\ntrain = train.drop(columns = \"row_id\")\n\ndel cv2_train, cv2_valid\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.201274,"end_time":"2020-11-21T09:37:33.88433","exception":false,"start_time":"2020-11-21T09:37:33.683056","status":"completed"},"tags":[]},"cell_type":"markdown","source":"In the previous version, I made a function for all the merges, fillna's and label encoding below. However, after adding a few features in this version, I ran into memory issues. I seemed as if a copy of train was kept in RAM at least temporarily, and I ran into an out-of memory error. Therefore, I unfortunately went back to an ugly version of code repetition for those steps (same code for validation and test_df)."},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:37:34.345597Z","iopub.status.busy":"2020-11-21T09:37:34.340525Z","iopub.status.idle":"2020-11-21T09:38:32.203429Z","shell.execute_reply":"2020-11-21T09:38:32.204093Z"},"papermill":{"duration":58.113526,"end_time":"2020-11-21T09:38:32.204289","exception":false,"start_time":"2020-11-21T09:37:34.090763","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"label_enc = LabelEncoder()\n\ntrain = train.merge(user_df, on = \"user_id\", how = \"left\") # get user_mean & watches_lectures\ntrain = train.merge(content_df, on = \"content_id\", how = \"left\") # get content_questions & content_mean\ntrain['content_questions'].fillna(0, inplace = True)\ntrain['content_mean'].fillna(0.5, inplace = True)\ntrain['watches_lecture'].fillna(0, inplace = True)\ntrain['user_questions'].fillna(0, inplace = True)\ntrain['user_mean'].fillna(0.5, inplace = True)\ntrain['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\ntrain['prior_question_had_explanation'].fillna(False, inplace = True)\ntrain['prior_question_had_explanation'] = label_enc.fit_transform(train['prior_question_had_explanation'])\ntrain[['content_questions', 'user_questions']] = train[['content_questions', 'user_questions']].astype(int)\ntrain.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:38:32.471812Z","iopub.status.busy":"2020-11-21T09:38:32.471038Z","iopub.status.idle":"2020-11-21T09:38:33.747528Z","shell.execute_reply":"2020-11-21T09:38:33.746792Z"},"papermill":{"duration":1.415509,"end_time":"2020-11-21T09:38:33.747649","exception":false,"start_time":"2020-11-21T09:38:32.33214","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"validation = validation.merge(user_df, on = \"user_id\", how = \"left\")\nvalidation = validation.merge(content_df, on = \"content_id\", how = \"left\")\nvalidation['content_questions'].fillna(0, inplace = True)\nvalidation['content_mean'].fillna(0.5, inplace = True)\nvalidation['watches_lecture'].fillna(0, inplace = True)\nvalidation['user_questions'].fillna(0, inplace = True)\nvalidation['user_mean'].fillna(0.5, inplace = True)\nvalidation['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\nvalidation['prior_question_had_explanation'].fillna(False, inplace = True)\nvalidation['prior_question_had_explanation'] = label_enc.fit_transform(validation['prior_question_had_explanation'])\nvalidation[['content_questions', 'user_questions']] = validation[['content_questions', 'user_questions']].astype(int)\nvalidation.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.129719,"end_time":"2020-11-21T09:38:34.006079","exception":false,"start_time":"2020-11-21T09:38:33.87636","status":"completed"},"tags":[]},"cell_type":"markdown","source":"As you can see below, I am using only 5 features for this baseline. Initially, I also included \"watches_lecture\" and \"prior_question_had_explanation\", but these two features are very questionable. With those two features, CV barely goes up (less than 0.001), the public score goes down 0.001 and feature importance is very low for both features. Therefore, I believe that the simpler model is preferred. If you want still check what the numbers look like with those features, you can do that by simply hashing in and out just a few lines (\"features\" below, and lines with \"categorical feature\" in the lgb.Dataset lines)."},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:38:34.27458Z","iopub.status.busy":"2020-11-21T09:38:34.273705Z","iopub.status.idle":"2020-11-21T09:38:42.177924Z","shell.execute_reply":"2020-11-21T09:38:42.177239Z"},"papermill":{"duration":8.043575,"end_time":"2020-11-21T09:38:42.17807","exception":false,"start_time":"2020-11-21T09:38:34.134495","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# features = ['user_questions', 'user_mean', 'content_questions', 'content_mean', 'watches_lecture',\n#             'prior_question_elapsed_time', 'prior_question_had_explanation']\n\nfeatures = ['user_questions', 'user_mean', 'content_questions', 'content_mean', 'prior_question_elapsed_time']\n\n\n#for now just taking 10.000.000 rows for training\ntrain = train.sample(n=10000000, random_state = 1)\n\ny_train = train['answered_correctly']\ntrain = train[features]\n\ny_val = validation['answered_correctly']\nvalidation = validation[features]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:51:47.70402Z","iopub.status.busy":"2020-11-21T09:51:47.702892Z","iopub.status.idle":"2020-11-21T09:51:47.706524Z","shell.execute_reply":"2020-11-21T09:51:47.705738Z"},"papermill":{"duration":0.143599,"end_time":"2020-11-21T09:51:47.706655","exception":false,"start_time":"2020-11-21T09:51:47.563056","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"env = riiideducation.make_env()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:51:48.047087Z","iopub.status.busy":"2020-11-21T09:51:48.046092Z","iopub.status.idle":"2020-11-21T09:51:48.049331Z","shell.execute_reply":"2020-11-21T09:51:48.048535Z"},"papermill":{"duration":0.207135,"end_time":"2020-11-21T09:51:48.049458","exception":false,"start_time":"2020-11-21T09:51:47.842323","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"iter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-21T09:51:48.340445Z","iopub.status.busy":"2020-11-21T09:51:48.339556Z","iopub.status.idle":"2020-11-21T09:51:48.739623Z","shell.execute_reply":"2020-11-21T09:51:48.740391Z"},"papermill":{"duration":0.556737,"end_time":"2020-11-21T09:51:48.740591","exception":false,"start_time":"2020-11-21T09:51:48.183854","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    test_df = test_df.merge(user_df, on = \"user_id\", how = \"left\")\n    test_df = test_df.merge(content_df, on = \"content_id\", how = \"left\")\n    test_df['content_questions'].fillna(0, inplace = True)\n    test_df['content_mean'].fillna(0.5, inplace = True)\n    test_df['watches_lecture'].fillna(0, inplace = True)\n    test_df['user_questions'].fillna(0, inplace = True)\n    test_df['user_mean'].fillna(0.5, inplace = True)\n    test_df['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\n    test_df['prior_question_had_explanation'].fillna(False, inplace = True)\n    test_df['prior_question_had_explanation'] = label_enc.fit_transform(test_df['prior_question_had_explanation'])\n    test_df[['content_questions', 'user_questions']] = test_df[['content_questions', 'user_questions']].astype(int)\n    test_df['answered_correctly'] =  model.predict(test_df[features])\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 2.2 Yuu modeling"},{"metadata":{"trusted":true},"cell_type":"code","source":"#this clears everything loaded in RAM, including the libraries\n%reset -f","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport riiideducation\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.style as style\nstyle.use('fivethirtyeight')\nimport seaborn as sns\nimport os\nimport lightgbm as lgb\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import LabelEncoder\nimport gc\nimport sys\npd.set_option('display.max_rows', None)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ncols_to_load = ['row_id', 'user_id', 'answered_correctly', 'content_id', 'prior_question_had_explanation', 'prior_question_elapsed_time']\ntrain = pd.read_pickle(\"../input/riiid-train-data-multiple-formats/riiid_train.pkl.gzip\")[cols_to_load]\ntrain['prior_question_had_explanation'] = train['prior_question_had_explanation'].astype('bool')\n\nprint(\"Train size:\", train.shape)\n\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nexample_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\nexample_sample_submission = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n# adding user features\n# PK: user_id + user_questions\n# 1. remove lecture content, leaving only questions\n# 2. for each user, count how questions answered the average accuracy for them\n# 3. Join back train_df to add if user watches_lecture before\n\nuser_df = train[train.answered_correctly != -1].groupby('user_id').agg({'answered_correctly': ['count', 'mean']}).reset_index()\nuser_df.columns = ['user_id', 'user_questions', 'user_mean']\n\nuser_lect = train.groupby([\"user_id\", \"answered_correctly\"]).size().unstack()\nuser_lect.columns = ['Lecture', 'Wrong', 'Right']\nuser_lect['Lecture'] = user_lect['Lecture'].fillna(0)\nuser_lect = user_lect.astype('Int64')\nuser_lect['watches_lecture'] = np.where(user_lect.Lecture > 0, 1, 0)\nuser_lect = user_lect.reset_index()\nuser_lect = user_lect[['user_id', 'watches_lecture']]\n\nuser_df = user_df.merge(user_lect, on = \"user_id\", how = \"left\")\ndel user_lect\nuser_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n# adding content features\n# PK: content_id\n# 1. remove lecture content, leaving only questions\n# 2. count how many people answered the question & the average accuracy of each question\n\ncontent_df = train[train.answered_correctly != -1].groupby('content_id').agg(\n    {'answered_correctly': ['count', 'mean']}).reset_index()\ncontent_df.columns = ['content_id', 'content_questions', 'content_mean']\ncontent_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n#using one of the validation sets composed by tito\ncv2_train = pd.read_pickle(\"../input/riiid-cross-validation-files/cv2_train.pickle\")['row_id']\ncv2_valid = pd.read_pickle(\"../input/riiid-cross-validation-files/cv2_valid.pickle\")['row_id']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train.answered_correctly != -1]\n\n#save mean before splitting\n#please be aware that there is an issues with train.prior_question_elapsed_time.mean()\n#see https://www.kaggle.com/c/riiid-test-answer-prediction/discussion/195032\nmean_prior = train.prior_question_elapsed_time.astype(\"float64\").mean()\n\nvalidation = train[train.row_id.isin(cv2_valid)]\ntrain = train[train.row_id.isin(cv2_train)]\n\nvalidation = validation.drop(columns = \"row_id\")\ntrain = train.drop(columns = \"row_id\")\n\ndel cv2_train, cv2_valid\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_enc = LabelEncoder()\n\ntrain = train.merge(user_df, on = \"user_id\", how = \"left\")\ntrain = train.merge(content_df, on = \"content_id\", how = \"left\")\ntrain['content_questions'].fillna(0, inplace = True)\ntrain['content_mean'].fillna(0.5, inplace = True)\ntrain['watches_lecture'].fillna(0, inplace = True)\ntrain['user_questions'].fillna(0, inplace = True)\ntrain['user_mean'].fillna(0.5, inplace = True)\ntrain['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\ntrain['prior_question_had_explanation'].fillna(False, inplace = True)\ntrain['prior_question_had_explanation'] = label_enc.fit_transform(train['prior_question_had_explanation'])\ntrain[['content_questions', 'user_questions']] = train[['content_questions', 'user_questions']].astype(int)\ntrain.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"validation = validation.merge(user_df, on = \"user_id\", how = \"left\")\nvalidation = validation.merge(content_df, on = \"content_id\", how = \"left\")\nvalidation['content_questions'].fillna(0, inplace = True)\nvalidation['content_mean'].fillna(0.5, inplace = True)\nvalidation['watches_lecture'].fillna(0, inplace = True)\nvalidation['user_questions'].fillna(0, inplace = True)\nvalidation['user_mean'].fillna(0.5, inplace = True)\nvalidation['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\nvalidation['prior_question_had_explanation'].fillna(False, inplace = True)\nvalidation['prior_question_had_explanation'] = label_enc.fit_transform(validation['prior_question_had_explanation'])\nvalidation[['content_questions', 'user_questions']] = validation[['content_questions', 'user_questions']].astype(int)\nvalidation.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features = ['user_questions', 'user_mean', 'content_questions', 'content_mean', 'watches_lecture',\n            'prior_question_elapsed_time', 'prior_question_had_explanation']\n\n# features = ['user_questions', 'user_mean', 'content_questions', 'content_mean', 'prior_question_elapsed_time']\n\n\n#for now just taking 1,000,000 rows for training\ntrain_original = train.copy()\ntrain = train_original.sample(n=5000000, random_state = 1)\n\ny_train = train['answered_correctly']\ntrain = train[features]\n\ny_val = validation['answered_correctly']\nvalidation = validation[features]\n\nprint(y_val.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Yuu modeling\n\nimport time\n\nimport xgboost as xgb\nfrom sklearn.metrics import mean_squared_error\nimport pandas as pd\nimport numpy as np\n\ndata_dmatrix = xgb.DMatrix(data=train,label=y_train)\n\nparams = {\"objective\":\"binary:logistic\",'colsample_bytree': 0.3,'learning_rate': 0.1,\n                'max_depth': 5, 'alpha': 10}\n\nxgb_clf = xgb.XGBClassifier(**params)\n\ncv_results = xgb.cv(dtrain=data_dmatrix, params=params, nfold=3,\n                    num_boost_round=50,early_stopping_rounds=10,metrics=\"logloss\", as_pandas=True, seed=123)\n\nstart_time = time.time()\nxgb_clf.fit(train, y_train)\n\nprint('Training Take: {} seconds'.format(time.time()-start_time))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# save model \n\n# pd.read_pickle(\"../input/riiid-train-data-multiple-formats/riiid_train.pkl.gzip\")[cols_to_load]\n\n# save_dir = '../input/'\n# xgb_clf.save_model(save_dir + 'xgb.bin')\n\nimport pickle\n\nwith open('../input/xgb.pkl', 'wb') as f:\n    pickle.dump(xgb_clf, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"help()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"metrics_df = pd.DataFrame({\n    'train_loss': cv_results['train-logloss-mean'], \n    'test_loss': cv_results['test-logloss-mean']\n})\n\nmetrics_df.plot.density()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nxgb_pred = xgb_clf.predict(train)\n\nprint(classification_report(y_train, xgb_pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xgb_pred_val = xgb_clf.predict(validation)\n\nprint(classification_report(y_val, xgb_pred_val))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_imp_df = pd.DataFrame({\n    'feature': features,\n    'importance': xgb_clf.feature_importances_\n})\nfeature_imp_df.index = features\n\nfeature_imp_df = feature_imp_df.sort_values('importance')\nfeature_imp_df.plot.bar()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# different sample size\n\nfor sample_size in [5000, 10000, 50000, 100000, 500000, 1000000, 5000000, 10000000]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import riiideducation\n\nenv = riiideducation.make_env()\niter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    test_df = test_df.merge(user_df, on = \"user_id\", how = \"left\")\n    test_df = test_df.merge(content_df, on = \"content_id\", how = \"left\")\n    \n    test_df['user_questions'].fillna(0, inplace = True)\n    test_df['user_mean'].fillna(0.5, inplace = True)\n    test_df['content_questions'].fillna(0, inplace = True)\n    test_df['content_mean'].fillna(0.5, inplace = True)\n    test_df['watches_lecture'].fillna(0, inplace = True)\n    test_df['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\n    test_df['prior_question_had_explanation'].fillna(False, inplace = True)\n    test_df['prior_question_had_explanation'] = label_enc.fit_transform(test_df['prior_question_had_explanation'])\n    test_df[['content_questions', 'user_questions']] = test_df[['content_questions', 'user_questions']].astype(int)\n    test_df['answered_correctly'] =  xgb_clf.predict(test_df[features])\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# all_test = []\n# for test_df, sample_prediction_df in iter_test:\n#     all_test.append(test_df)\n#     all_test.append(sample_prediction_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"all_test[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}