{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-01-01T14:52:24.037701Z","iopub.status.busy":"2021-01-01T14:52:24.037016Z","iopub.status.idle":"2021-01-01T14:52:54.1926Z","shell.execute_reply":"2021-01-01T14:52:54.19202Z"},"papermill":{"duration":30.229826,"end_time":"2021-01-01T14:52:54.192713","exception":false,"start_time":"2021-01-01T14:52:23.962887","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"!pip install ../input/python-datatable/datatable-0.11.0-cp37-cp37m-manylinux2010_x86_64.whl > /dev/null 2>&1","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:52:54.332391Z","iopub.status.busy":"2021-01-01T14:52:54.33159Z","iopub.status.idle":"2021-01-01T14:52:54.473261Z","shell.execute_reply":"2021-01-01T14:52:54.472614Z"},"papermill":{"duration":0.213872,"end_time":"2021-01-01T14:52:54.473381","exception":false,"start_time":"2021-01-01T14:52:54.259509","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import numpy as np\nimport random\nimport pandas as pd\nimport joblib","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2021-01-01T14:52:54.61437Z","iopub.status.busy":"2021-01-01T14:52:54.613464Z","iopub.status.idle":"2021-01-01T14:52:55.456565Z","shell.execute_reply":"2021-01-01T14:52:55.456014Z"},"papermill":{"duration":0.915032,"end_time":"2021-01-01T14:52:55.456678","exception":false,"start_time":"2021-01-01T14:52:54.541646","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"from collections import defaultdict\nimport datatable as dt\nimport lightgbm as lgb\nfrom matplotlib import pyplot as plt\nimport riiideducation\nfrom sklearn.metrics import roc_auc_score\nimport gc\n\n_ = np.seterr(divide='ignore', invalid='ignore')","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.06781,"end_time":"2021-01-01T14:52:55.593134","exception":false,"start_time":"2021-01-01T14:52:55.525324","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Preprocess"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:52:55.734982Z","iopub.status.busy":"2021-01-01T14:52:55.73393Z","iopub.status.idle":"2021-01-01T14:52:55.737411Z","shell.execute_reply":"2021-01-01T14:52:55.736745Z"},"papermill":{"duration":0.076393,"end_time":"2021-01-01T14:52:55.737559","exception":false,"start_time":"2021-01-01T14:52:55.661166","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"data_types_dict = {\n    'timestamp': 'int64',\n    'user_id': 'int32', \n    'content_id': 'int16', \n    'content_type_id':'int8', \n    'task_container_id': 'int16',\n    #'user_answer': 'int8',\n    'answered_correctly': 'int8', \n    'prior_question_elapsed_time': 'float32', \n    'prior_question_had_explanation': 'bool'\n}\ntarget = 'answered_correctly'","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:52:55.945094Z","iopub.status.busy":"2021-01-01T14:52:55.944248Z","iopub.status.idle":"2021-01-01T14:54:31.526165Z","shell.execute_reply":"2021-01-01T14:54:31.525623Z"},"papermill":{"duration":95.687499,"end_time":"2021-01-01T14:54:31.526277","exception":false,"start_time":"2021-01-01T14:52:55.838778","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('start read train data...')\ntrain_df = dt.fread('../input/riiid-test-answer-prediction/train.csv', columns=set(data_types_dict.keys())).to_pandas()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:31.831305Z","iopub.status.busy":"2021-01-01T14:54:31.830589Z","iopub.status.idle":"2021-01-01T14:54:31.835079Z","shell.execute_reply":"2021-01-01T14:54:31.834371Z"},"papermill":{"duration":0.081165,"end_time":"2021-01-01T14:54:31.835182","exception":false,"start_time":"2021-01-01T14:54:31.754017","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('start handle lecture data...')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:31.991996Z","iopub.status.busy":"2021-01-01T14:54:31.991377Z","iopub.status.idle":"2021-01-01T14:54:32.005197Z","shell.execute_reply":"2021-01-01T14:54:32.004548Z"},"papermill":{"duration":0.098098,"end_time":"2021-01-01T14:54:32.005327","exception":false,"start_time":"2021-01-01T14:54:31.907229","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"#reading in lecture df\nlectures_df = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:32.180159Z","iopub.status.busy":"2021-01-01T14:54:32.162074Z","iopub.status.idle":"2021-01-01T14:54:32.189828Z","shell.execute_reply":"2021-01-01T14:54:32.189262Z"},"papermill":{"duration":0.109897,"end_time":"2021-01-01T14:54:32.189944","exception":false,"start_time":"2021-01-01T14:54:32.080047","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"lectures_df['type_of'] = lectures_df['type_of'].replace('solving question', 'solving_question')\n\nlectures_df = pd.get_dummies(lectures_df, columns=['part', 'type_of'])\n\npart_lectures_columns = [column for column in lectures_df.columns if column.startswith('part')]\n\ntypes_of_lectures_columns = [column for column in lectures_df.columns if column.startswith('type_of_')]","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:32.342561Z","iopub.status.busy":"2021-01-01T14:54:32.341845Z","iopub.status.idle":"2021-01-01T14:54:33.402826Z","shell.execute_reply":"2021-01-01T14:54:33.402189Z"},"papermill":{"duration":1.141155,"end_time":"2021-01-01T14:54:33.402957","exception":false,"start_time":"2021-01-01T14:54:32.261802","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_lectures = train_df[train_df.content_type_id == True].merge(lectures_df, left_on='content_id', right_on='lecture_id', how='left')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:33.556128Z","iopub.status.busy":"2021-01-01T14:54:33.555402Z","iopub.status.idle":"2021-01-01T14:54:34.0921Z","shell.execute_reply":"2021-01-01T14:54:34.091588Z"},"papermill":{"duration":0.617718,"end_time":"2021-01-01T14:54:34.092219","exception":false,"start_time":"2021-01-01T14:54:33.474501","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_lecture_stats_part = train_lectures.groupby('user_id',as_index = False)[part_lectures_columns + types_of_lectures_columns].sum()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:34.246259Z","iopub.status.busy":"2021-01-01T14:54:34.245505Z","iopub.status.idle":"2021-01-01T14:54:34.255843Z","shell.execute_reply":"2021-01-01T14:54:34.25513Z"},"papermill":{"duration":0.090179,"end_time":"2021-01-01T14:54:34.255974","exception":false,"start_time":"2021-01-01T14:54:34.165795","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"lecturedata_types_dict = {   \n    'user_id': 'int32', \n    'part_1': 'int8',\n    'part_2': 'int8',\n    'part_3': 'int8',\n    'part_4': 'int8',\n    'part_5': 'int8',\n    'part_6': 'int8',\n    'part_7': 'int8',\n    'type_of_concept': 'int8',\n    'type_of_intention': 'int8',\n    'type_of_solving_question': 'int8',\n    'type_of_starter': 'int8'\n}\nuser_lecture_stats_part = user_lecture_stats_part.astype(lecturedata_types_dict)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:34.446606Z","iopub.status.busy":"2021-01-01T14:54:34.445961Z","iopub.status.idle":"2021-01-01T14:54:34.454107Z","shell.execute_reply":"2021-01-01T14:54:34.454747Z"},"papermill":{"duration":0.094389,"end_time":"2021-01-01T14:54:34.454984","exception":false,"start_time":"2021-01-01T14:54:34.360595","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"for column in user_lecture_stats_part.columns:\n    #bool_column = column + '_boolean'\n    if(column !='user_id'):\n        user_lecture_stats_part[column] = (user_lecture_stats_part[column] > 0).astype('int8')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:34.613766Z","iopub.status.busy":"2021-01-01T14:54:34.612928Z","iopub.status.idle":"2021-01-01T14:54:34.619052Z","shell.execute_reply":"2021-01-01T14:54:34.618305Z"},"papermill":{"duration":0.087114,"end_time":"2021-01-01T14:54:34.619156","exception":false,"start_time":"2021-01-01T14:54:34.532042","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_lecture_stats_part.dtypes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:34.843911Z","iopub.status.busy":"2021-01-01T14:54:34.843111Z","iopub.status.idle":"2021-01-01T14:54:34.849071Z","shell.execute_reply":"2021-01-01T14:54:34.848326Z"},"papermill":{"duration":0.156963,"end_time":"2021-01-01T14:54:34.849184","exception":false,"start_time":"2021-01-01T14:54:34.692221","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"#clearing memory\ndel(train_lectures)\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:35.00232Z","iopub.status.busy":"2021-01-01T14:54:35.001729Z","iopub.status.idle":"2021-01-01T14:54:39.478263Z","shell.execute_reply":"2021-01-01T14:54:39.477738Z"},"papermill":{"duration":4.555228,"end_time":"2021-01-01T14:54:39.478374","exception":false,"start_time":"2021-01-01T14:54:34.923146","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_lecture_agg = train_df.groupby('user_id')['content_type_id'].agg(['sum', 'count'])\nuser_lecture_agg=user_lecture_agg.astype('int16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:39.629537Z","iopub.status.busy":"2021-01-01T14:54:39.628858Z","iopub.status.idle":"2021-01-01T14:54:50.765837Z","shell.execute_reply":"2021-01-01T14:54:50.765283Z"},"papermill":{"duration":11.216076,"end_time":"2021-01-01T14:54:50.765949","exception":false,"start_time":"2021-01-01T14:54:39.549873","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\n#1= if the event was the user watching a lecture.\ncum = train_df.groupby('user_id')['content_type_id'].agg(['cumsum', 'cumcount'])\ncum['cumcount']=cum['cumcount']+1\ntrain_df['user_interaction_count'] = cum['cumcount'] \ntrain_df['user_interaction_timestamp_mean'] = train_df['timestamp']/cum['cumcount'] \ntrain_df['user_lecture_sum'] = cum['cumsum'] \ntrain_df['user_lecture_lv'] = cum['cumsum'] / cum['cumcount']\n\n\ntrain_df.user_lecture_lv=train_df.user_lecture_lv.astype('float16')\ntrain_df.user_lecture_sum=train_df.user_lecture_sum.astype('int16')\ntrain_df.user_interaction_count=train_df.user_interaction_count.astype('int16')\ntrain_df['user_interaction_timestamp_mean']=train_df['user_interaction_timestamp_mean']/(1000*3600)\ntrain_df.user_interaction_timestamp_mean=train_df.user_interaction_timestamp_mean.astype('float32')\n","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:50.912805Z","iopub.status.busy":"2021-01-01T14:54:50.912071Z","iopub.status.idle":"2021-01-01T14:54:50.915539Z","shell.execute_reply":"2021-01-01T14:54:50.916021Z"},"papermill":{"duration":0.078718,"end_time":"2021-01-01T14:54:50.916145","exception":false,"start_time":"2021-01-01T14:54:50.837427","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"#pd.options.display.max_rows = 200","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:51.178636Z","iopub.status.busy":"2021-01-01T14:54:51.177749Z","iopub.status.idle":"2021-01-01T14:54:51.18382Z","shell.execute_reply":"2021-01-01T14:54:51.183189Z"},"papermill":{"duration":0.195983,"end_time":"2021-01-01T14:54:51.183928","exception":false,"start_time":"2021-01-01T14:54:50.987945","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del cum\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:51.33213Z","iopub.status.busy":"2021-01-01T14:54:51.331379Z","iopub.status.idle":"2021-01-01T14:54:51.33495Z","shell.execute_reply":"2021-01-01T14:54:51.334399Z"},"papermill":{"duration":0.079153,"end_time":"2021-01-01T14:54:51.335049","exception":false,"start_time":"2021-01-01T14:54:51.255896","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('start handle train_df...')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:54:56.268338Z","iopub.status.busy":"2021-01-01T14:54:56.2676Z","iopub.status.idle":"2021-01-01T14:55:10.080965Z","shell.execute_reply":"2021-01-01T14:55:10.080361Z"},"papermill":{"duration":18.674319,"end_time":"2021-01-01T14:55:10.081072","exception":false,"start_time":"2021-01-01T14:54:51.406753","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['prior_question_had_explanation'].fillna(False, inplace=True)\ntrain_df = train_df.astype(data_types_dict)\ntrain_df = train_df[train_df[target] != -1].reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:55:10.800254Z","iopub.status.busy":"2021-01-01T14:55:10.799392Z","iopub.status.idle":"2021-01-01T14:55:16.72745Z","shell.execute_reply":"2021-01-01T14:55:16.727946Z"},"papermill":{"duration":6.573236,"end_time":"2021-01-01T14:55:16.728091","exception":false,"start_time":"2021-01-01T14:55:10.154855","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"content_explation_agg=train_df[[\"content_id\",\"prior_question_had_explanation\",target]].groupby([\"content_id\",\"prior_question_had_explanation\"])[target].agg(['mean'])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:55:16.882174Z","iopub.status.busy":"2021-01-01T14:55:16.881171Z","iopub.status.idle":"2021-01-01T14:55:16.886049Z","shell.execute_reply":"2021-01-01T14:55:16.886598Z"},"papermill":{"duration":0.084604,"end_time":"2021-01-01T14:55:16.88673","exception":false,"start_time":"2021-01-01T14:55:16.802126","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"content_explation_agg.dtypes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:55:17.042078Z","iopub.status.busy":"2021-01-01T14:55:17.041407Z","iopub.status.idle":"2021-01-01T14:55:17.062901Z","shell.execute_reply":"2021-01-01T14:55:17.06215Z"},"papermill":{"duration":0.100864,"end_time":"2021-01-01T14:55:17.063033","exception":false,"start_time":"2021-01-01T14:55:16.962169","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"content_explation_agg=content_explation_agg.unstack()\n\ncontent_explation_agg=content_explation_agg.reset_index()\ncontent_explation_agg.columns = ['content_id', 'content_explation_false_mean','content_explation_true_mean']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"content_explation_agg['explation_impact'] = ((content_explation_agg['content_explation_true_mean'] - \n                                             content_explation_agg['content_explation_false_mean'])/\n                                             content_explation_agg['content_explation_false_mean'])\n\ncontent_explation_agg['explation_impact'] = np.where(content_explation_agg['explation_impact'] < 0, 0, content_explation_agg['explation_impact'])\n\ncontent_explation_agg['explation_impact'] = np.where(content_explation_agg['content_explation_false_mean']==0, content_explation_agg['content_explation_true_mean'], content_explation_agg['explation_impact'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"content_explation_agg.content_id=content_explation_agg.content_id.astype('int16')\ncontent_explation_agg.content_explation_false_mean=content_explation_agg.content_explation_false_mean.astype('float16')\ncontent_explation_agg.content_explation_true_mean=content_explation_agg.content_explation_true_mean.astype('float16')\ncontent_explation_agg.explation_impact = content_explation_agg.explation_impact.astype('float16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:55:17.447756Z","iopub.status.busy":"2021-01-01T14:55:17.447129Z","iopub.status.idle":"2021-01-01T14:55:17.451216Z","shell.execute_reply":"2021-01-01T14:55:17.450351Z"},"papermill":{"duration":0.08214,"end_time":"2021-01-01T14:55:17.451367","exception":false,"start_time":"2021-01-01T14:55:17.369227","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('start handle attempt_no...')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:55:17.611728Z","iopub.status.busy":"2021-01-01T14:55:17.610874Z","iopub.status.idle":"2021-01-01T14:57:32.074203Z","shell.execute_reply":"2021-01-01T14:57:32.073153Z"},"papermill":{"duration":134.54821,"end_time":"2021-01-01T14:57:32.074567","exception":false,"start_time":"2021-01-01T14:55:17.526357","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\ntrain_df[\"attempt_no\"] = 1\ntrain_df.attempt_no=train_df.attempt_no.astype('int8')\nattempt_no_agg=train_df.groupby([\"user_id\",\"content_id\"])[\"attempt_no\"].agg(['sum']).astype('int8')\ntrain_df[\"attempt_no\"] = train_df[[\"user_id\",\"content_id\",'attempt_no']].groupby([\"user_id\",\"content_id\"])[\"attempt_no\"].cumsum()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:32.23465Z","iopub.status.busy":"2021-01-01T14:57:32.233748Z","iopub.status.idle":"2021-01-01T14:57:32.598731Z","shell.execute_reply":"2021-01-01T14:57:32.597823Z"},"papermill":{"duration":0.445595,"end_time":"2021-01-01T14:57:32.598911","exception":false,"start_time":"2021-01-01T14:57:32.153316","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"attempt_no_agg=attempt_no_agg[attempt_no_agg['sum'] >1]","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:32.926772Z","iopub.status.busy":"2021-01-01T14:57:32.92576Z","iopub.status.idle":"2021-01-01T14:57:33.191476Z","shell.execute_reply":"2021-01-01T14:57:33.190915Z"},"papermill":{"duration":0.386866,"end_time":"2021-01-01T14:57:33.191592","exception":false,"start_time":"2021-01-01T14:57:32.804726","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('start handle timestamp...')\nprior_question_elapsed_time_mean=train_df['prior_question_elapsed_time'].mean()\ntrain_df['prior_question_elapsed_time'].fillna(prior_question_elapsed_time_mean, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:34.663079Z","iopub.status.busy":"2021-01-01T14:57:34.662332Z","iopub.status.idle":"2021-01-01T14:57:37.305887Z","shell.execute_reply":"2021-01-01T14:57:37.305306Z"},"papermill":{"duration":4.038367,"end_time":"2021-01-01T14:57:37.306008","exception":false,"start_time":"2021-01-01T14:57:33.267641","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"max_timestamp_u = train_df[['user_id','timestamp']].groupby(['user_id']).agg(['max']).reset_index()\nmax_timestamp_u.columns = ['user_id', 'max_time_stamp']\nmax_timestamp_u.user_id=max_timestamp_u.user_id.astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:37.465652Z","iopub.status.busy":"2021-01-01T14:57:37.46498Z","iopub.status.idle":"2021-01-01T14:57:45.075081Z","shell.execute_reply":"2021-01-01T14:57:45.074377Z"},"papermill":{"duration":7.69414,"end_time":"2021-01-01T14:57:45.075195","exception":false,"start_time":"2021-01-01T14:57:37.381055","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\ntrain_df['lagtime'] = train_df.groupby('user_id')['timestamp'].shift()\n\nmax_timestamp_u2 = train_df[['user_id','lagtime']].groupby(['user_id']).agg(['max']).reset_index()\nmax_timestamp_u2.columns = ['user_id', 'max_time_stamp2']\nmax_timestamp_u2.user_id=max_timestamp_u2.user_id.astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:45.246081Z","iopub.status.busy":"2021-01-01T14:57:45.244865Z","iopub.status.idle":"2021-01-01T14:57:45.811415Z","shell.execute_reply":"2021-01-01T14:57:45.810867Z"},"papermill":{"duration":0.660769,"end_time":"2021-01-01T14:57:45.811551","exception":false,"start_time":"2021-01-01T14:57:45.150782","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['lagtime']=train_df['timestamp']-train_df['lagtime']\nlagtime_mean=train_df['lagtime'].mean()\ntrain_df['lagtime'].fillna(lagtime_mean, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:45.969982Z","iopub.status.busy":"2021-01-01T14:57:45.969043Z","iopub.status.idle":"2021-01-01T14:57:46.442288Z","shell.execute_reply":"2021-01-01T14:57:46.441747Z"},"papermill":{"duration":0.55514,"end_time":"2021-01-01T14:57:46.442403","exception":false,"start_time":"2021-01-01T14:57:45.887263","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['lagtime']=train_df['lagtime']/(1000*3600)\ntrain_df.lagtime=train_df.lagtime.astype('float32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:46.763048Z","iopub.status.busy":"2021-01-01T14:57:46.761316Z","iopub.status.idle":"2021-01-01T14:57:55.351373Z","shell.execute_reply":"2021-01-01T14:57:55.350685Z"},"papermill":{"duration":8.673254,"end_time":"2021-01-01T14:57:55.351515","exception":false,"start_time":"2021-01-01T14:57:46.678261","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['lagtime2'] = train_df.groupby('user_id')['timestamp'].shift(2)\n\nmax_timestamp_u3 = train_df[['user_id','lagtime2']].groupby(['user_id']).agg(['max']).reset_index()\nmax_timestamp_u3.columns = ['user_id', 'max_time_stamp3']\nmax_timestamp_u3.user_id=max_timestamp_u3.user_id.astype('int32')\n\ntrain_df['lagtime2']=train_df['timestamp']-train_df['lagtime2']\nlagtime_mean2=train_df['lagtime2'].mean()\ntrain_df['lagtime2'].fillna(lagtime_mean2, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:55.518542Z","iopub.status.busy":"2021-01-01T14:57:55.517636Z","iopub.status.idle":"2021-01-01T14:57:55.971108Z","shell.execute_reply":"2021-01-01T14:57:55.970218Z"},"papermill":{"duration":0.544643,"end_time":"2021-01-01T14:57:55.971241","exception":false,"start_time":"2021-01-01T14:57:55.426598","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['lagtime2']=train_df['lagtime2']/(1000*3600)\ntrain_df.lagtime2=train_df.lagtime2.astype('float32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:57:56.205722Z","iopub.status.busy":"2021-01-01T14:57:56.204991Z","iopub.status.idle":"2021-01-01T14:58:00.988077Z","shell.execute_reply":"2021-01-01T14:58:00.987473Z"},"papermill":{"duration":4.903556,"end_time":"2021-01-01T14:58:00.988188","exception":false,"start_time":"2021-01-01T14:57:56.084632","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['lagtime3'] = train_df.groupby('user_id')['timestamp'].shift(3)\n\ntrain_df['lagtime3']=train_df['timestamp']-train_df['lagtime3']\nlagtime_mean3=train_df['lagtime3'].mean()\ntrain_df['lagtime3'].fillna(lagtime_mean3, inplace=True)\ntrain_df['lagtime3']=train_df['lagtime3']/(1000*3600)\ntrain_df.lagtime3=train_df.lagtime3.astype('float32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:01.307534Z","iopub.status.busy":"2021-01-01T14:58:01.306635Z","iopub.status.idle":"2021-01-01T14:58:03.147249Z","shell.execute_reply":"2021-01-01T14:58:03.146578Z"},"papermill":{"duration":1.924293,"end_time":"2021-01-01T14:58:03.147359","exception":false,"start_time":"2021-01-01T14:58:01.223066","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['timestamp']=train_df['timestamp']/(1000*3600)\ntrain_df.timestamp=train_df.timestamp.astype('float16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:03.308561Z","iopub.status.busy":"2021-01-01T14:58:03.307915Z","iopub.status.idle":"2021-01-01T14:58:10.908689Z","shell.execute_reply":"2021-01-01T14:58:10.908085Z"},"papermill":{"duration":7.683837,"end_time":"2021-01-01T14:58:10.90881","exception":false,"start_time":"2021-01-01T14:58:03.224973","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_prior_question_elapsed_time = train_df[['user_id','prior_question_elapsed_time']].groupby(['user_id']).tail(1)\nuser_prior_question_elapsed_time.columns = ['user_id', 'prior_question_elapsed_time']","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:11.067031Z","iopub.status.busy":"2021-01-01T14:58:11.066366Z","iopub.status.idle":"2021-01-01T14:58:14.78055Z","shell.execute_reply":"2021-01-01T14:58:14.779977Z"},"papermill":{"duration":3.795845,"end_time":"2021-01-01T14:58:14.780665","exception":false,"start_time":"2021-01-01T14:58:10.98482","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['delta_prior_question_elapsed_time'] = train_df.groupby('user_id')['prior_question_elapsed_time'].shift()\ntrain_df['delta_prior_question_elapsed_time']=train_df['prior_question_elapsed_time']-train_df['delta_prior_question_elapsed_time']","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:14.938521Z","iopub.status.busy":"2021-01-01T14:58:14.937861Z","iopub.status.idle":"2021-01-01T14:58:15.375206Z","shell.execute_reply":"2021-01-01T14:58:15.37576Z"},"papermill":{"duration":0.519261,"end_time":"2021-01-01T14:58:15.375906","exception":false,"start_time":"2021-01-01T14:58:14.856645","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"delta_prior_question_elapsed_time_mean=train_df['delta_prior_question_elapsed_time'].mean()\ntrain_df['delta_prior_question_elapsed_time'].fillna(delta_prior_question_elapsed_time_mean, inplace=True)\ntrain_df.delta_prior_question_elapsed_time=train_df.delta_prior_question_elapsed_time.astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:15.539358Z","iopub.status.busy":"2021-01-01T14:58:15.53711Z","iopub.status.idle":"2021-01-01T14:58:41.183214Z","shell.execute_reply":"2021-01-01T14:58:41.183748Z"},"papermill":{"duration":25.730868,"end_time":"2021-01-01T14:58:41.183904","exception":false,"start_time":"2021-01-01T14:58:15.453036","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\ntrain_df['lag'] = train_df.groupby('user_id')[target].shift()\n\ncum = train_df.groupby('user_id')['lag'].agg(['cumsum', 'cumcount'])\nuser_agg = train_df.groupby('user_id')['lag'].agg(['sum', 'count']).astype('int16')\ncum['cumsum'].fillna(0, inplace=True)\n\ntrain_df['user_correctness'] = cum['cumsum'] / cum['cumcount']\ntrain_df['user_correct_count'] = cum['cumsum']\ntrain_df['user_uncorrect_count'] = cum['cumcount']-cum['cumsum']\ntrain_df.drop(columns=['lag'], inplace=True)\ntrain_df['user_correctness'].fillna(0.67, inplace=True)\ntrain_df.user_correctness=train_df.user_correctness.astype('float16')\ntrain_df.user_correct_count=train_df.user_correct_count.astype('int16')\ntrain_df.user_uncorrect_count=train_df.user_uncorrect_count.astype('int16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:41.462095Z","iopub.status.busy":"2021-01-01T14:58:41.343163Z","iopub.status.idle":"2021-01-01T14:58:41.467979Z","shell.execute_reply":"2021-01-01T14:58:41.467124Z"},"papermill":{"duration":0.20639,"end_time":"2021-01-01T14:58:41.468111","exception":false,"start_time":"2021-01-01T14:58:41.261721","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del cum\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:41.948034Z","iopub.status.busy":"2021-01-01T14:58:41.946872Z","iopub.status.idle":"2021-01-01T14:58:45.713302Z","shell.execute_reply":"2021-01-01T14:58:45.71258Z"},"papermill":{"duration":3.889922,"end_time":"2021-01-01T14:58:45.713466","exception":false,"start_time":"2021-01-01T14:58:41.823544","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df.prior_question_had_explanation=train_df.prior_question_had_explanation.astype('int8')\nexplanation_agg = train_df.groupby('user_id')['prior_question_had_explanation'].agg(['sum', 'count'])\nexplanation_agg=explanation_agg.astype('int16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:45.956053Z","iopub.status.busy":"2021-01-01T14:58:45.955382Z","iopub.status.idle":"2021-01-01T14:58:55.75437Z","shell.execute_reply":"2021-01-01T14:58:55.75381Z"},"papermill":{"duration":9.923319,"end_time":"2021-01-01T14:58:55.754513","exception":false,"start_time":"2021-01-01T14:58:45.831194","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\n\n#train_df['lag'] = train_df.groupby('user_id')['prior_question_had_explanation'].shift()\n\ncum = train_df.groupby('user_id')['prior_question_had_explanation'].agg(['cumsum', 'cumcount'])\ncum['cumcount']=cum['cumcount']+1\ntrain_df['explanation_mean'] = cum['cumsum'] / cum['cumcount']\ntrain_df['explanation_true_count'] = cum['cumsum'] \ntrain_df['explanation_false_count'] =  cum['cumcount']-cum['cumsum']\n#train_df.drop(columns=['lag'], inplace=True)\n\ntrain_df.explanation_mean=train_df.explanation_mean.astype('float16')\ntrain_df.explanation_true_count=train_df.explanation_true_count.astype('int16')\ntrain_df.explanation_false_count=train_df.explanation_false_count.astype('int16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:55.982963Z","iopub.status.busy":"2021-01-01T14:58:55.982094Z","iopub.status.idle":"2021-01-01T14:58:55.987887Z","shell.execute_reply":"2021-01-01T14:58:55.98838Z"},"papermill":{"duration":0.155763,"end_time":"2021-01-01T14:58:55.98855","exception":false,"start_time":"2021-01-01T14:58:55.832787","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del cum\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:58:56.149781Z","iopub.status.busy":"2021-01-01T14:58:56.149139Z","iopub.status.idle":"2021-01-01T14:59:09.592886Z","shell.execute_reply":"2021-01-01T14:59:09.592172Z"},"papermill":{"duration":13.526908,"end_time":"2021-01-01T14:59:09.592999","exception":false,"start_time":"2021-01-01T14:58:56.066091","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"content_agg = train_df.groupby('content_id')[target].agg(['sum', 'count','var'])\ntask_container_agg = train_df.groupby('task_container_id')[target].agg(['sum', 'count','var'])\ncontent_agg=content_agg.astype('float32')\ntask_container_agg=task_container_agg.astype('float32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:09.757379Z","iopub.status.busy":"2021-01-01T14:59:09.756692Z","iopub.status.idle":"2021-01-01T14:59:14.68875Z","shell.execute_reply":"2021-01-01T14:59:14.688061Z"},"papermill":{"duration":5.018025,"end_time":"2021-01-01T14:59:14.688868","exception":false,"start_time":"2021-01-01T14:59:09.670843","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['task_container_uncor_count'] = train_df['task_container_id'].map(task_container_agg['count']-task_container_agg['sum']).astype('int32')\ntrain_df['task_container_cor_count'] = train_df['task_container_id'].map(task_container_agg['sum']).astype('int32')\ntrain_df['task_container_std'] = train_df['task_container_id'].map(task_container_agg['var']).astype('float16')\ntrain_df['task_container_correctness'] = train_df['task_container_id'].map(task_container_agg['sum'] / task_container_agg['count'])\ntrain_df.task_container_correctness=train_df.task_container_correctness.astype('float16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:14.853507Z","iopub.status.busy":"2021-01-01T14:59:14.852799Z","iopub.status.idle":"2021-01-01T14:59:20.515942Z","shell.execute_reply":"2021-01-01T14:59:20.515244Z"},"papermill":{"duration":5.748572,"end_time":"2021-01-01T14:59:20.516053","exception":false,"start_time":"2021-01-01T14:59:14.767481","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"content_elapsed_time_agg=train_df.groupby('content_id')['prior_question_elapsed_time'].agg(['mean'])\ncontent_had_explanation_agg=train_df.groupby('content_id')['prior_question_had_explanation'].agg(['mean'])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:20.680282Z","iopub.status.busy":"2021-01-01T14:59:20.679078Z","iopub.status.idle":"2021-01-01T14:59:20.684134Z","shell.execute_reply":"2021-01-01T14:59:20.683272Z"},"papermill":{"duration":0.09009,"end_time":"2021-01-01T14:59:20.684281","exception":false,"start_time":"2021-01-01T14:59:20.594191","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df.dtypes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:20.931997Z","iopub.status.busy":"2021-01-01T14:59:20.928941Z","iopub.status.idle":"2021-01-01T14:59:20.964457Z","shell.execute_reply":"2021-01-01T14:59:20.963886Z"},"papermill":{"duration":0.160541,"end_time":"2021-01-01T14:59:20.964566","exception":false,"start_time":"2021-01-01T14:59:20.804025","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:21.127579Z","iopub.status.busy":"2021-01-01T14:59:21.126776Z","iopub.status.idle":"2021-01-01T14:59:21.131897Z","shell.execute_reply":"2021-01-01T14:59:21.132373Z"},"papermill":{"duration":0.088457,"end_time":"2021-01-01T14:59:21.132527","exception":false,"start_time":"2021-01-01T14:59:21.04407","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"print('start questions data...')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:21.305871Z","iopub.status.busy":"2021-01-01T14:59:21.305168Z","iopub.status.idle":"2021-01-01T14:59:21.322928Z","shell.execute_reply":"2021-01-01T14:59:21.322315Z"},"papermill":{"duration":0.11021,"end_time":"2021-01-01T14:59:21.323045","exception":false,"start_time":"2021-01-01T14:59:21.212835","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df = pd.read_csv(\n    '../input/riiid-test-answer-prediction/questions.csv', \n    usecols=[0, 1,3,4],\n    dtype={'question_id': 'int16','bundle_id': 'int16', 'part': 'int8','tags': 'str'}\n)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:21.491906Z","iopub.status.busy":"2021-01-01T14:59:21.490997Z","iopub.status.idle":"2021-01-01T14:59:21.502002Z","shell.execute_reply":"2021-01-01T14:59:21.50111Z"},"papermill":{"duration":0.097802,"end_time":"2021-01-01T14:59:21.502156","exception":false,"start_time":"2021-01-01T14:59:21.404354","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"bundle_agg = questions_df.groupby('bundle_id')['question_id'].agg(['count'])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:21.751573Z","iopub.status.busy":"2021-01-01T14:59:21.750919Z","iopub.status.idle":"2021-01-01T14:59:21.756908Z","shell.execute_reply":"2021-01-01T14:59:21.757396Z"},"papermill":{"duration":0.134238,"end_time":"2021-01-01T14:59:21.757555","exception":false,"start_time":"2021-01-01T14:59:21.623317","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df['content_sub_bundle'] = questions_df['bundle_id'].map(bundle_agg['count']).astype('int8')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:21.929082Z","iopub.status.busy":"2021-01-01T14:59:21.92786Z","iopub.status.idle":"2021-01-01T14:59:21.933621Z","shell.execute_reply":"2021-01-01T14:59:21.932899Z"},"papermill":{"duration":0.094742,"end_time":"2021-01-01T14:59:21.933739","exception":false,"start_time":"2021-01-01T14:59:21.838997","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\nquestions_df['tags'].fillna('188', inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:22.102424Z","iopub.status.busy":"2021-01-01T14:59:22.10128Z","iopub.status.idle":"2021-01-01T14:59:22.10526Z","shell.execute_reply":"2021-01-01T14:59:22.104572Z"},"papermill":{"duration":0.091252,"end_time":"2021-01-01T14:59:22.105401","exception":false,"start_time":"2021-01-01T14:59:22.014149","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def gettags(tags,num):\n    tags_splits=tags.split(\" \")\n    result='' \n    for t in tags_splits:\n        x=int(t)\n        if(x<32*(num+1) and x>=32*num):#num \n            result=result+' '+t\n    return result","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:22.369321Z","iopub.status.busy":"2021-01-01T14:59:22.364343Z","iopub.status.idle":"2021-01-01T14:59:22.569981Z","shell.execute_reply":"2021-01-01T14:59:22.569279Z"},"papermill":{"duration":0.341346,"end_time":"2021-01-01T14:59:22.570098","exception":false,"start_time":"2021-01-01T14:59:22.228752","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfor num in range(0,6):\n    questions_df[\"tags\"+str(num)] = questions_df[\"tags\"].apply(lambda row: gettags(row,num))\n    le = LabelEncoder()\n    le.fit(np.unique(questions_df['tags'+str(num)].values))\n    #questions_df[['tags'+str(num)]=\n    questions_df['tags'+str(num)]=questions_df[['tags'+str(num)]].apply(le.transform)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:22.801312Z","iopub.status.busy":"2021-01-01T14:59:22.800044Z","iopub.status.idle":"2021-01-01T14:59:22.802642Z","shell.execute_reply":"2021-01-01T14:59:22.803712Z"},"papermill":{"duration":0.129959,"end_time":"2021-01-01T14:59:22.803952","exception":false,"start_time":"2021-01-01T14:59:22.673993","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df_dict = {   \n    'tags0': 'int8',\n    'tags1': 'int8',\n    'tags2': 'int8',\n    'tags3': 'int8',\n    'tags4': 'int8',\n    'tags5': 'int8',\n    #'tags6': 'int8',\n    #'tags7': 'int8'\n}\nquestions_df = questions_df.astype(questions_df_dict)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:22.988914Z","iopub.status.busy":"2021-01-01T14:59:22.987947Z","iopub.status.idle":"2021-01-01T14:59:22.991396Z","shell.execute_reply":"2021-01-01T14:59:22.990488Z"},"papermill":{"duration":0.095024,"end_time":"2021-01-01T14:59:22.991577","exception":false,"start_time":"2021-01-01T14:59:22.896553","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df.drop(columns=['tags'], inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:23.169152Z","iopub.status.busy":"2021-01-01T14:59:23.168277Z","iopub.status.idle":"2021-01-01T14:59:23.171652Z","shell.execute_reply":"2021-01-01T14:59:23.170926Z"},"papermill":{"duration":0.093531,"end_time":"2021-01-01T14:59:23.171799","exception":false,"start_time":"2021-01-01T14:59:23.078268","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\nquestions_df['part_bundle_id']=questions_df['part']*100000+questions_df['bundle_id']\nquestions_df.part_bundle_id=questions_df.part_bundle_id.astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:23.706604Z","iopub.status.busy":"2021-01-01T14:59:23.705588Z","iopub.status.idle":"2021-01-01T14:59:23.709396Z","shell.execute_reply":"2021-01-01T14:59:23.708553Z"},"papermill":{"duration":0.107517,"end_time":"2021-01-01T14:59:23.709613","exception":false,"start_time":"2021-01-01T14:59:23.602096","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df.rename(columns={'question_id':'content_id'}, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:23.887647Z","iopub.status.busy":"2021-01-01T14:59:23.88628Z","iopub.status.idle":"2021-01-01T14:59:23.910233Z","shell.execute_reply":"2021-01-01T14:59:23.90969Z"},"papermill":{"duration":0.114822,"end_time":"2021-01-01T14:59:23.910339","exception":false,"start_time":"2021-01-01T14:59:23.795517","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df = pd.merge(questions_df, content_explation_agg, on='content_id', how='left',right_index=True)#","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:24.078053Z","iopub.status.busy":"2021-01-01T14:59:24.077256Z","iopub.status.idle":"2021-01-01T14:59:24.081266Z","shell.execute_reply":"2021-01-01T14:59:24.080662Z"},"papermill":{"duration":0.089048,"end_time":"2021-01-01T14:59:24.081379","exception":false,"start_time":"2021-01-01T14:59:23.992331","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del content_explation_agg","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:24.25524Z","iopub.status.busy":"2021-01-01T14:59:24.254562Z","iopub.status.idle":"2021-01-01T14:59:24.267219Z","shell.execute_reply":"2021-01-01T14:59:24.266637Z"},"papermill":{"duration":0.103882,"end_time":"2021-01-01T14:59:24.267348","exception":false,"start_time":"2021-01-01T14:59:24.163466","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df['content_correctness'] = questions_df['content_id'].map(content_agg['sum'] / content_agg['count'])\nquestions_df.content_correctness=questions_df.content_correctness.astype('float16')\nquestions_df['content_correctness_std'] = questions_df['content_id'].map(content_agg['var'])\nquestions_df.content_correctness_std=questions_df.content_correctness_std.astype('float16')\nquestions_df['content_uncorrect_count'] = questions_df['content_id'].map(content_agg['count']-content_agg['sum']).astype('int32')\nquestions_df['content_correct_count'] = questions_df['content_id'].map(content_agg['sum']).astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:24.439911Z","iopub.status.busy":"2021-01-01T14:59:24.439194Z","iopub.status.idle":"2021-01-01T14:59:24.447135Z","shell.execute_reply":"2021-01-01T14:59:24.446527Z"},"papermill":{"duration":0.09535,"end_time":"2021-01-01T14:59:24.447248","exception":false,"start_time":"2021-01-01T14:59:24.351898","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df['content_elapsed_time_mean'] = questions_df['content_id'].map(content_elapsed_time_agg['mean'])\nquestions_df.content_elapsed_time_mean=questions_df.content_elapsed_time_mean.astype('float16')\nquestions_df['content_had_explanation_mean'] = questions_df['content_id'].map(content_had_explanation_agg['mean'])\nquestions_df.content_had_explanation_mean=questions_df.content_had_explanation_mean.astype('float16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:24.684489Z","iopub.status.busy":"2021-01-01T14:59:24.68374Z","iopub.status.idle":"2021-01-01T14:59:24.689222Z","shell.execute_reply":"2021-01-01T14:59:24.688383Z"},"papermill":{"duration":0.158177,"end_time":"2021-01-01T14:59:24.689363","exception":false,"start_time":"2021-01-01T14:59:24.531186","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del content_elapsed_time_agg\ndel content_had_explanation_agg\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:24.944788Z","iopub.status.busy":"2021-01-01T14:59:24.944118Z","iopub.status.idle":"2021-01-01T14:59:24.954399Z","shell.execute_reply":"2021-01-01T14:59:24.953811Z"},"papermill":{"duration":0.141038,"end_time":"2021-01-01T14:59:24.954541","exception":false,"start_time":"2021-01-01T14:59:24.813503","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"part_agg = questions_df.groupby('part')['content_correctness'].agg(['mean', 'var'])\nquestions_df['part_correctness_mean'] = questions_df['part'].map(part_agg['mean'])\nquestions_df['part_correctness_std'] = questions_df['part'].map(part_agg['var'])\nquestions_df.part_correctness_mean=questions_df.part_correctness_mean.astype('float16')\nquestions_df.part_correctness_std=questions_df.part_correctness_std.astype('float16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:25.135134Z","iopub.status.busy":"2021-01-01T14:59:25.134487Z","iopub.status.idle":"2021-01-01T14:59:25.151381Z","shell.execute_reply":"2021-01-01T14:59:25.150682Z"},"papermill":{"duration":0.110705,"end_time":"2021-01-01T14:59:25.151547","exception":false,"start_time":"2021-01-01T14:59:25.040842","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"part_agg = questions_df.groupby('part')['content_uncorrect_count'].agg(['sum'])\nquestions_df['part_uncor_count'] = questions_df['part'].map(part_agg['sum']).astype('int32')\n#\npart_agg = questions_df.groupby('part')['content_correct_count'].agg(['sum'])\nquestions_df['part_cor_count'] = questions_df['part'].map(part_agg['sum']).astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:25.409418Z","iopub.status.busy":"2021-01-01T14:59:25.408759Z","iopub.status.idle":"2021-01-01T14:59:25.420254Z","shell.execute_reply":"2021-01-01T14:59:25.419385Z"},"papermill":{"duration":0.14219,"end_time":"2021-01-01T14:59:25.420401","exception":false,"start_time":"2021-01-01T14:59:25.278211","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"bundle_agg = questions_df.groupby('bundle_id')['content_correctness'].agg(['mean'])\nquestions_df['bundle_correctness_mean'] = questions_df['bundle_id'].map(bundle_agg['mean'])\nquestions_df.bundle_correctness_mean=questions_df.bundle_correctness_mean.astype('float16')","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:26.220002Z","iopub.status.busy":"2021-01-01T14:59:26.219224Z","iopub.status.idle":"2021-01-01T14:59:26.225821Z","shell.execute_reply":"2021-01-01T14:59:26.224736Z"},"papermill":{"duration":0.142636,"end_time":"2021-01-01T14:59:26.225996","exception":false,"start_time":"2021-01-01T14:59:26.08336","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"questions_df.dtypes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:26.46802Z","iopub.status.busy":"2021-01-01T14:59:26.467354Z","iopub.status.idle":"2021-01-01T14:59:26.472736Z","shell.execute_reply":"2021-01-01T14:59:26.472016Z"},"papermill":{"duration":0.160228,"end_time":"2021-01-01T14:59:26.472858","exception":false,"start_time":"2021-01-01T14:59:26.31263","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del content_agg\ndel bundle_agg\ndel part_agg\n#del tags1_agg\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:27.1847Z","iopub.status.busy":"2021-01-01T14:59:27.183857Z","iopub.status.idle":"2021-01-01T14:59:27.190574Z","shell.execute_reply":"2021-01-01T14:59:27.190004Z"},"papermill":{"duration":0.098917,"end_time":"2021-01-01T14:59:27.190692","exception":false,"start_time":"2021-01-01T14:59:27.091775","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"len(train_df)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.085151,"end_time":"2021-01-01T14:59:27.547165","exception":false,"start_time":"2021-01-01T14:59:27.462014","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Train"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:27.764094Z","iopub.status.busy":"2021-01-01T14:59:27.750776Z","iopub.status.idle":"2021-01-01T14:59:27.77194Z","shell.execute_reply":"2021-01-01T14:59:27.772646Z"},"papermill":{"duration":0.140659,"end_time":"2021-01-01T14:59:27.772809","exception":false,"start_time":"2021-01-01T14:59:27.63215","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"features_dict = {\n    'timestamp':'float16',#\n    'user_interaction_count':'int16',\n    'user_interaction_timestamp_mean':'float32',\n    'lagtime':'float32',#\n    'lagtime2':'float32',\n    'lagtime3':'float32',\n    'content_id':'int16',\n    'task_container_id':'int16',\n    'user_lecture_sum':'int16',#\n    'user_lecture_lv':'float16',##\n    'prior_question_elapsed_time':'float32',#\n    'delta_prior_question_elapsed_time':'int32',#\n    'user_correctness':'float16',#\n    'user_uncorrect_count':'int16',#\n    'user_correct_count':'int16',#\n    'content_correctness_std':'float16',\n    'content_correct_count':'int32',\n    'content_uncorrect_count':'int32',#\n    'content_elapsed_time_mean':'float16',\n    'content_had_explanation_mean':'float16',\n    'content_explation_false_mean':'float16',\n    'content_explation_true_mean':'float16',\n    #\n    'explation_impact':'float16',\n    #\n    'task_container_correctness':'float16',\n    'task_container_std':'float16',\n    'task_container_cor_count':'int32',#\n    'task_container_uncor_count':'int32',#\n    'attempt_no':'int8',#\n    'part':'int8',\n    'part_correctness_mean':'float16',\n    'part_correctness_std':'float16',\n    'part_uncor_count':'int32',\n    'part_cor_count':'int32',\n    'tags0': 'int8',\n    'tags1': 'int8',\n    'tags2': 'int8',\n    'tags3': 'int8',\n    'tags4': 'int8',\n    'tags5': 'int8',\n    'part_bundle_id':'int32',\n    'content_sub_bundle':'int8',\n    'prior_question_had_explanation':'int8',\n    'explanation_mean':'float16', #\n    'explanation_false_count':'int16',#\n    'explanation_true_count':'int16'\n}\ncategorical_columns= [\n    'content_id',\n    'task_container_id',\n    'part',\n    'tags0',\n    'tags1',\n    'tags2',\n    'tags3',\n    'tags4',\n    'tags5',\n    'part_bundle_id',\n    'content_sub_bundle',\n    'prior_question_had_explanation'\n]\n\nfeatures=list(features_dict.keys())\n","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:28.016714Z","iopub.status.busy":"2021-01-01T14:59:28.015718Z","iopub.status.idle":"2021-01-01T14:59:28.018179Z","shell.execute_reply":"2021-01-01T14:59:28.019096Z"},"papermill":{"duration":0.126649,"end_time":"2021-01-01T14:59:28.019311","exception":false,"start_time":"2021-01-01T14:59:27.892662","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"flag_lgbm=True\nclfs = list()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:28.309256Z","iopub.status.busy":"2021-01-01T14:59:28.308368Z","iopub.status.idle":"2021-01-01T14:59:28.311563Z","shell.execute_reply":"2021-01-01T14:59:28.312139Z"},"papermill":{"duration":0.1443,"end_time":"2021-01-01T14:59:28.312266","exception":false,"start_time":"2021-01-01T14:59:28.167966","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"params = {\n        'num_leaves': 355,\n        'max_bin': 256+128+64,\n        'feature_fraction': 0.548,\n        'bagging_fraction': 0.648,\n        'bagging_freq': 11,\n        'min_data_in_leaf': 31+64+8,\n        'max_depth': 15+4,\n        'objective': 'binary',\n        'learning_rate': 0.0212345,\n        \"boosting_type\": \"gbdt\",\n        \"metric\": 'auc',\n        'seed': 123\n        #'boost_from_average': False\n}","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:28.51764Z","iopub.status.busy":"2021-01-01T14:59:28.516706Z","iopub.status.idle":"2021-01-01T14:59:39.877053Z","shell.execute_reply":"2021-01-01T14:59:39.876423Z"},"papermill":{"duration":11.476331,"end_time":"2021-01-01T14:59:39.877164","exception":false,"start_time":"2021-01-01T14:59:28.400833","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"trains=list()\nvalids=list()\n\nnum=1\n\nfor i in range(0,num):\n    \n    train_df_clf=train_df[1200*10000:2*1200*10000]\n    print('sample end')\n       \n    del train_df\n    \n    users=train_df_clf['user_id'].drop_duplicates()#\n    users=users.sample(frac=0.08, random_state=42)\n    users_df=pd.DataFrame()\n    users_df['user_id']=users.values\n   \n    valid_df_newuser = pd.merge(train_df_clf, users_df, on=['user_id'], how='inner',right_index=True)\n    del users_df\n    del users\n    gc.collect()\n    train_df_clf.drop(valid_df_newuser.index, inplace=True)\n    print('pd.merge(train_df_clf, questions_df)')\n\n    train_df_clf = pd.merge(train_df_clf, questions_df, on='content_id', how='left',right_index=True)#\n    valid_df_newuser = pd.merge(valid_df_newuser, questions_df, on='content_id', how='left',right_index=True)#\n\n    print('valid_df')\n    valid_df = train_df_clf.sample(frac=0.03, random_state=42)\n    train_df_clf.drop(valid_df.index, inplace=True)\n   \n    valid_df = valid_df.append(valid_df_newuser)\n    del valid_df_newuser\n    gc.collect()\n\n    trains.append(train_df_clf)\n    valids.append(valid_df)\n    print('train_df_clf length：',len(train_df_clf))\n    print('valid_df length：',len(valid_df))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:40.172653Z","iopub.status.busy":"2021-01-01T14:59:40.171738Z","iopub.status.idle":"2021-01-01T14:59:40.179017Z","shell.execute_reply":"2021-01-01T14:59:40.178326Z"},"papermill":{"duration":0.21349,"end_time":"2021-01-01T14:59:40.179132","exception":false,"start_time":"2021-01-01T14:59:39.965642","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del train_df_clf\ndel valid_df\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T14:59:40.368488Z","iopub.status.busy":"2021-01-01T14:59:40.367577Z","iopub.status.idle":"2021-01-01T15:32:18.074914Z","shell.execute_reply":"2021-01-01T15:32:18.075527Z"},"papermill":{"duration":1957.806289,"end_time":"2021-01-01T15:32:18.075707","exception":false,"start_time":"2021-01-01T14:59:40.269418","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\nfor i in range(0,num):\n\n    #Don't use DF to create lightgbm dataset, rather use np array:\n    X_train_np = trains[i][features].values.astype(np.float32)\n    X_valid_np = valids[i][features].values.astype(np.float32)\n    #features = train.columns\n    tr_data = lgb.Dataset(X_train_np, label=trains[i][target], feature_name=list(features))\n    va_data = lgb.Dataset(X_valid_np, label=valids[i][target], feature_name=list(features))\n    \n    del trains\n    del valids\n    del X_train_np\n    del X_valid_np\n    gc.collect()\n\n    model = lgb.train(\n        params, \n        tr_data,\n        num_boost_round=7000,\n        valid_sets=[tr_data, va_data],\n        early_stopping_rounds=50,\n        feature_name=features,\n        categorical_feature=categorical_columns,\n        verbose_eval=50\n    )\n    clfs.append(model)\n\n    fig,ax = plt.subplots(figsize=(15,15))\n    lgb.plot_importance(model, ax=ax,importance_type='gain',max_num_features=50)\n    plt.show()\n\n    del tr_data\n    del va_data\n    gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"MAX_SEQ = 240 # 210\nACCEPTED_USER_CONTENT_SIZE = 2 # 2\nEMBED_SIZE = 256 # 256\nBATCH_SIZE = 64+32 # 96\nDROPOUT = 0.1 # 0.1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class FFN(nn.Module):\n    def __init__(self, state_size = 200, forward_expansion = 1, bn_size = MAX_SEQ - 1, dropout=0.2):\n        super(FFN, self).__init__()\n        self.state_size = state_size\n        \n        self.lr1 = nn.Linear(state_size, forward_expansion * state_size)\n        self.relu = nn.ReLU()\n        self.bn = nn.BatchNorm1d(bn_size)\n        self.lr2 = nn.Linear(forward_expansion * state_size, state_size)\n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, x):\n        x = self.relu(self.lr1(x))\n        x = self.bn(x)\n        x = self.lr2(x)\n        return self.dropout(x)\n    \nclass FFN0(nn.Module):\n    def __init__(self, state_size = 200, forward_expansion = 1, bn_size = MAX_SEQ - 1, dropout=0.2):\n        super(FFN0, self).__init__()\n        self.state_size = state_size\n\n        self.lr1 = nn.Linear(state_size, forward_expansion * state_size)\n        self.relu = nn.ReLU()\n        self.lr2 = nn.Linear(forward_expansion * state_size, state_size)\n        self.layer_normal = nn.LayerNorm(state_size) \n        self.dropout = nn.Dropout(0.2)\n    \n    def forward(self, x):\n        x = self.lr1(x)\n        x = self.relu(x)\n        x = self.lr2(x)\n        x=self.layer_normal(x)\n        return self.dropout(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def future_mask(seq_length):\n    future_mask = (np.triu(np.ones([seq_length, seq_length]), k = 1)).astype('bool')\n    return torch.from_numpy(future_mask)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class TransformerBlock(nn.Module):\n    def __init__(self, embed_dim, heads = 8, dropout = DROPOUT, forward_expansion = 1):\n        super(TransformerBlock, self).__init__()\n        self.multi_att = nn.MultiheadAttention(embed_dim=embed_dim, num_heads=heads, dropout=dropout)\n        self.dropout = nn.Dropout(dropout)\n        self.layer_normal = nn.LayerNorm(embed_dim)\n        self.ffn = FFN(embed_dim, forward_expansion = forward_expansion, dropout=dropout)\n        self.ffn0  = FFN0(embed_dim, forward_expansion = forward_expansion, dropout=dropout)\n        self.layer_normal_2 = nn.LayerNorm(embed_dim)\n\n    def forward(self, value, key, query, att_mask):\n        att_output, att_weight = self.multi_att(value, key, query, attn_mask=att_mask)\n        att_output = self.dropout(self.layer_normal(att_output + value))\n        att_output = att_output.permute(1, 0, 2) # att_output: [s_len, bs, embed] => [bs, s_len, embed]\n        x = self.ffn(att_output)\n        x1 = self.ffn0(att_output)\n        x = self.dropout(self.layer_normal_2(x + x1 + att_output))\n        return x.squeeze(-1), att_weight\n    \nclass Encoder(nn.Module):\n    def __init__(self, n_skill, max_seq=100, embed_dim=128, dropout = DROPOUT, forward_expansion = 1, num_layers=1, heads = 8):\n        super(Encoder, self).__init__()\n        self.n_skill, self.embed_dim = n_skill, embed_dim\n        self.embedding = nn.Embedding(2 * n_skill + 1, embed_dim)\n        self.pos_embedding = nn.Embedding(max_seq - 1, embed_dim)\n        self.e_embedding = nn.Embedding(n_skill+1, embed_dim)\n        self.layers = nn.ModuleList([TransformerBlock(embed_dim, forward_expansion = forward_expansion) for _ in range(num_layers)])\n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, x, question_ids):\n        device = x.device\n        x = self.embedding(x)\n        pos_id = torch.arange(x.size(1)).unsqueeze(0).to(device)\n        pos_x = self.pos_embedding(pos_id)\n        x = self.dropout(x + pos_x)\n        x = x.permute(1, 0, 2) # x: [bs, s_len, embed] => [s_len, bs, embed]\n        e = self.e_embedding(question_ids)\n        e = e.permute(1, 0, 2)\n        for layer in self.layers:\n            att_mask = future_mask(e.size(0)).to(device)\n            x, att_weight = layer(e, x, x, att_mask=att_mask)\n            x = x.permute(1, 0, 2)\n        x = x.permute(1, 0, 2)\n        return x, att_weight\n\nclass SAKTModel(nn.Module):\n    def __init__(self, n_skill, max_seq = MAX_SEQ, embed_dim=128, dropout = DROPOUT, forward_expansion = 1, enc_layers=1, heads = 8):\n        super(SAKTModel, self).__init__()\n        self.encoder = Encoder(n_skill, max_seq, embed_dim, dropout, forward_expansion, num_layers=enc_layers)\n        self.pred = nn.Linear(embed_dim, 1)\n        \n    def forward(self, x, question_ids):\n        x, att_weight = self.encoder(x, question_ids)\n        x = self.pred(x)\n        return x.squeeze(-1), att_weight","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"skills = joblib.load(\"/kaggle/input/riiid-sakt-model-dataset-public/skills.pkl.zip\")\ngroup = joblib.load(\"/kaggle/input/riiid-sakt-model-dataset-public/group.pkl.zip\")\n# m_path = \"/kaggle/input/riiid-sakt-model-dataset-public/sakt_model.pt\"\n\n# skills = joblib.load(\"/kaggle/input/fork-of-riiid-sakt-model-full/skills.pkl.zip\")\n# group = joblib.load(\"/kaggle/input/fork-of-riiid-sakt-model-full/group.pkl.zip\")\nm_path = \"/kaggle/input/v4-fork-of-riiid-sakt-model-full/sakt_model.pt\"\n\nn_skill = len(skills)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nnn_model = SAKTModel(n_skill, embed_dim = EMBED_SIZE)\n\ntry:\n    nn_model.load_state_dict(torch.load(m_path))\nexcept:\n    nn_model.load_state_dict(torch.load(m_path, map_location='cpu'))\n\nnn_model.to(device)\nnn_model.eval()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.097021,"end_time":"2021-01-01T15:32:35.303776","exception":false,"start_time":"2021-01-01T15:32:35.206755","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Inference"},{"metadata":{"trusted":true},"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, samples, test_df, n_skill, max_seq = MAX_SEQ):\n        super(TestDataset, self).__init__()\n        self.samples, self.user_ids, self.test_df = samples, [x for x in test_df[\"user_id\"].unique()], test_df\n        self.n_skill, self.max_seq = n_skill, max_seq\n\n    def __len__(self):\n        return self.test_df.shape[0]\n    \n    def __getitem__(self, index):\n        test_info = self.test_df.iloc[index]\n        \n        user_id = test_info['user_id']\n        target_id = test_info['content_id']\n        \n        content_id_seq = np.zeros(self.max_seq, dtype=int)\n        answered_correctly_seq = np.zeros(self.max_seq, dtype=int)\n        \n        if user_id in self.samples.index:\n            content_id, answered_correctly = self.samples[user_id]\n            \n            seq_len = len(content_id)\n            \n            if seq_len >= self.max_seq:\n                content_id_seq = content_id[-self.max_seq:]\n                answered_correctly_seq = answered_correctly[-self.max_seq:]\n            else:\n                content_id_seq[-seq_len:] = content_id\n                answered_correctly_seq[-seq_len:] = answered_correctly\n                \n        x = content_id_seq[1:].copy()\n        x += (answered_correctly_seq[1:] == 1) * self.n_skill\n        \n        questions = np.append(content_id_seq[2:], [target_id])\n        \n        return x, questions","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:32:35.509418Z","iopub.status.busy":"2021-01-01T15:32:35.504066Z","iopub.status.idle":"2021-01-01T15:32:35.52081Z","shell.execute_reply":"2021-01-01T15:32:35.520108Z"},"papermill":{"duration":0.121359,"end_time":"2021-01-01T15:32:35.520924","exception":false,"start_time":"2021-01-01T15:32:35.399565","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# class TestDataset(Dataset):\n#     def __init__(self, samples, test_df, skills, max_seq=MAX_SEQ): \n#         super(TestDataset, self).__init__()\n#         self.samples = samples\n#         self.user_ids = [x for x in test_df[\"user_id\"].unique()]\n#         self.test_df = test_df\n#         self.skills = skills\n#         self.n_skill = len(skills)\n#         self.max_seq = max_seq\n\n#     def __len__(self):\n#         return self.test_df.shape[0]\n\n#     def __getitem__(self, index):\n#         test_info = self.test_df.iloc[index]\n\n#         user_id = test_info[\"user_id\"]\n#         target_id = test_info[\"content_id\"]\n\n#         q = np.zeros(self.max_seq, dtype=int)\n#         qa = np.zeros(self.max_seq, dtype=int)\n\n#         if user_id in self.samples.index:\n#             q_, qa_ = self.samples[user_id]\n            \n#             seq_len = len(q_)\n\n#             if seq_len >= self.max_seq:\n#                 q = q_[-self.max_seq:]\n#                 qa = qa_[-self.max_seq:]\n#             else:\n#                 q[-seq_len:] = q_\n#                 qa[-seq_len:] = qa_          \n        \n#         x = np.zeros(self.max_seq-1, dtype=int)\n#         x = q[1:].copy()\n#         x += (qa[1:] == 1) * self.n_skill\n        \n#         questions = np.append(q[2:], [target_id])\n        \n#         return x, questions","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:32:35.721885Z","iopub.status.busy":"2021-01-01T15:32:35.721225Z","iopub.status.idle":"2021-01-01T15:32:36.186631Z","shell.execute_reply":"2021-01-01T15:32:36.185962Z"},"papermill":{"duration":0.56857,"end_time":"2021-01-01T15:32:36.186743","exception":false,"start_time":"2021-01-01T15:32:35.618173","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_sum_dict = user_agg['sum'].astype('int16').to_dict(defaultdict(int))\nuser_count_dict = user_agg['count'].astype('int16').to_dict(defaultdict(int))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:32:36.480963Z","iopub.status.busy":"2021-01-01T15:32:36.476566Z","iopub.status.idle":"2021-01-01T15:32:36.89138Z","shell.execute_reply":"2021-01-01T15:32:36.890672Z"},"papermill":{"duration":0.607959,"end_time":"2021-01-01T15:32:36.891548","exception":false,"start_time":"2021-01-01T15:32:36.283589","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"del user_agg\ngc.collect()\n\ntask_container_sum_dict = task_container_agg['sum'].astype('int32').to_dict(defaultdict(int))\ntask_container_count_dict = task_container_agg['count'].astype('int32').to_dict(defaultdict(int))\ntask_container_std_dict = task_container_agg['var'].astype('float16').to_dict(defaultdict(int))\n\nexplanation_sum_dict = explanation_agg['sum'].astype('int16').to_dict(defaultdict(int))\nexplanation_count_dict = explanation_agg['count'].astype('int16').to_dict(defaultdict(int))\ndel task_container_agg\ndel explanation_agg\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:32:37.459313Z","iopub.status.busy":"2021-01-01T15:32:37.458668Z","iopub.status.idle":"2021-01-01T15:32:38.187068Z","shell.execute_reply":"2021-01-01T15:32:38.186456Z"},"papermill":{"duration":0.830767,"end_time":"2021-01-01T15:32:38.18718","exception":false,"start_time":"2021-01-01T15:32:37.356413","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"user_lecture_sum_dict = user_lecture_agg['sum'].astype('int16').to_dict(defaultdict(int))\nuser_lecture_count_dict = user_lecture_agg['count'].astype('int16').to_dict(defaultdict(int))\n\ndel user_lecture_agg\n#del lagtime_agg\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:32:38.391723Z","iopub.status.busy":"2021-01-01T15:32:38.390679Z","iopub.status.idle":"2021-01-01T15:32:39.620196Z","shell.execute_reply":"2021-01-01T15:32:39.61952Z"},"papermill":{"duration":1.333591,"end_time":"2021-01-01T15:32:39.620307","exception":false,"start_time":"2021-01-01T15:32:38.286716","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"max_timestamp_u_dict=max_timestamp_u.set_index('user_id').to_dict()\nmax_timestamp_u_dict2=max_timestamp_u2.set_index('user_id').to_dict()\nmax_timestamp_u_dict3=max_timestamp_u3.set_index('user_id').to_dict()\nuser_prior_question_elapsed_time_dict=user_prior_question_elapsed_time.set_index('user_id').to_dict()\ndel max_timestamp_u\ndel max_timestamp_u2\ndel max_timestamp_u3\ndel user_prior_question_elapsed_time\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:32:39.823643Z","iopub.status.busy":"2021-01-01T15:32:39.822923Z","iopub.status.idle":"2021-01-01T15:33:06.067697Z","shell.execute_reply":"2021-01-01T15:33:06.067132Z"},"papermill":{"duration":26.349629,"end_time":"2021-01-01T15:33:06.067816","exception":false,"start_time":"2021-01-01T15:32:39.718187","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"attempt_no_sum_dict = attempt_no_agg['sum'].to_dict(defaultdict(int))\n\ndel attempt_no_agg\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:33:06.271389Z","iopub.status.busy":"2021-01-01T15:33:06.270564Z","iopub.status.idle":"2021-01-01T15:33:06.274265Z","shell.execute_reply":"2021-01-01T15:33:06.274782Z"},"papermill":{"duration":0.109024,"end_time":"2021-01-01T15:33:06.274915","exception":false,"start_time":"2021-01-01T15:33:06.165891","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def get_max_attempt(user_id,content_id):\n    k = (user_id,content_id)\n\n    if k in attempt_no_sum_dict.keys():\n        attempt_no_sum_dict[k]+=1\n        return attempt_no_sum_dict[k]\n\n    attempt_no_sum_dict[k] = 1\n    return attempt_no_sum_dict[k]","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:33:06.67873Z","iopub.status.busy":"2021-01-01T15:33:06.678056Z","iopub.status.idle":"2021-01-01T15:33:06.682008Z","shell.execute_reply":"2021-01-01T15:33:06.681408Z"},"papermill":{"duration":0.105496,"end_time":"2021-01-01T15:33:06.682134","exception":false,"start_time":"2021-01-01T15:33:06.576638","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:33:06.887457Z","iopub.status.busy":"2021-01-01T15:33:06.886566Z","iopub.status.idle":"2021-01-01T15:33:06.889937Z","shell.execute_reply":"2021-01-01T15:33:06.889197Z"},"papermill":{"duration":0.108095,"end_time":"2021-01-01T15:33:06.890075","exception":false,"start_time":"2021-01-01T15:33:06.78198","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"iter_test = env.iter_test()\nprior_test_df = None\nprev_test_df = None","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:33:07.18951Z","iopub.status.busy":"2021-01-01T15:33:07.188693Z","iopub.status.idle":"2021-01-01T15:33:07.191801Z","shell.execute_reply":"2021-01-01T15:33:07.190978Z"},"papermill":{"duration":0.153237,"end_time":"2021-01-01T15:33:07.191922","exception":false,"start_time":"2021-01-01T15:33:07.038685","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"N=[0.4,0.6]","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:33:07.521537Z","iopub.status.busy":"2021-01-01T15:33:07.511228Z","iopub.status.idle":"2021-01-01T15:33:08.591323Z","shell.execute_reply":"2021-01-01T15:33:08.591899Z"},"papermill":{"duration":1.252204,"end_time":"2021-01-01T15:33:08.592051","exception":false,"start_time":"2021-01-01T15:33:07.339847","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"%%time\n\nfor (test_df, sample_prediction_df) in iter_test:\n            \n    ##############################\n    ### SAKT v4\n\n    test_df1 = test_df.copy()\n    if (prev_test_df is not None):\n        prev_test_df['answered_correctly'] = eval(test_df1['prior_group_answers_correct'].iloc[0])\n        prev_test_df = prev_test_df[prev_test_df.content_type_id == False]\n        prev_group = prev_test_df[['user_id', 'content_id', 'answered_correctly']].groupby('user_id').apply(lambda r: (\n            r['content_id'].values,\n            r['answered_correctly'].values))\n        for prev_user_id in prev_group.index:\n            if prev_user_id in group.index:\n                group[prev_user_id] = (np.append(group[prev_user_id][0], prev_group[prev_user_id][0])[-MAX_SEQ:], \n                                       np.append(group[prev_user_id][1], prev_group[prev_user_id][1])[-MAX_SEQ:]\n                                      )\n            else:\n                group[prev_user_id] = (\n                    prev_group[prev_user_id][0], \n                    prev_group[prev_user_id][1]\n                )\n                \n    prev_test_df = test_df1.copy()\n    \n    test_df1 = test_df1[test_df1.content_type_id == False]\n    test_dataset = TestDataset(group, test_df1, n_skill, max_seq=MAX_SEQ)\n    test_dataloader = DataLoader(test_dataset, batch_size = 51200, shuffle=False)\n        \n    outs = []\n\n    for item in test_dataloader:\n        x = item[0].to(device).long()\n        target_id = item[1].to(device).long()\n        with torch.no_grad():\n            output, att_weight = nn_model(x, target_id)\n        outs.extend(torch.sigmoid(output)[:, -1].view(-1).data.cpu().numpy())\n    \n    ###########\n#     item = next(iter(test_dataloader))\n#     x = item[0].to(device).long()\n#     target_id = item[1].to(device).long()\n#     with torch.no_grad():\n#         output, _ = nn_model(x, target_id)   \n#     output = torch.sigmoid(output)\n#     output = output[:, -1]\n#     outs = output.cpu().numpy()\n    ###########\n    \n    #########################################\n    ######## for lgb\n    if prior_test_df is not None:\n        prior_test_df[target] = eval(test_df['prior_group_answers_correct'].iloc[0])\n        prior_test_df = prior_test_df[prior_test_df[target] != -1].reset_index(drop=True)\n        prior_test_df['prior_question_had_explanation'].fillna(False, inplace=True)       \n        prior_test_df.prior_question_had_explanation=prior_test_df.prior_question_had_explanation.astype('int8')\n    \n        user_ids = prior_test_df['user_id'].values\n        targets = prior_test_df[target].values        \n        \n        for user_id, answered_correctly in zip(user_ids,targets):\n            user_sum_dict[user_id] += answered_correctly\n            user_count_dict[user_id] += 1            \n\n    prior_test_df = test_df.copy() \n        \n    \n    question_len=len( test_df[test_df['content_type_id'] == 0])\n    test_df['prior_question_had_explanation'].fillna(False, inplace=True)\n    test_df.prior_question_had_explanation=test_df.prior_question_had_explanation.astype('int8')\n    test_df['prior_question_elapsed_time'].fillna(prior_question_elapsed_time_mean, inplace=True)\n    \n\n    user_lecture_sum = np.zeros(question_len, dtype=np.int16)\n    user_lecture_count = np.zeros(question_len, dtype=np.int16) \n    \n    user_sum = np.zeros(question_len, dtype=np.int16)\n    user_count = np.zeros(question_len, dtype=np.int16)\n\n    task_container_sum = np.zeros(question_len, dtype=np.int32)\n    task_container_count = np.zeros(question_len, dtype=np.int32)\n    task_container_std = np.zeros(question_len, dtype=np.float16)\n\n    explanation_sum = np.zeros(question_len, dtype=np.int32)\n    explanation_count = np.zeros(question_len, dtype=np.int32)\n    delta_prior_question_elapsed_time = np.zeros(question_len, dtype=np.int32)\n\n    attempt_no_count = np.zeros(question_len, dtype=np.int16)\n    lagtime = np.zeros(question_len, dtype=np.float32)\n    lagtime2 = np.zeros(question_len, dtype=np.float32)\n    lagtime3 = np.zeros(question_len, dtype=np.float32)\n    \n    i=0\n    for j, (user_id,prior_question_had_explanation,content_type_id,prior_question_elapsed_time,timestamp, content_id,task_container_id) in enumerate(zip(test_df['user_id'].values,test_df['prior_question_had_explanation'].values,test_df['content_type_id'].values,test_df['prior_question_elapsed_time'].values,test_df['timestamp'].values, test_df['content_id'].values, test_df['task_container_id'].values)):\n        \n         #\n        user_lecture_sum_dict[user_id] += content_type_id\n        user_lecture_count_dict[user_id] += 1\n        if(content_type_id==1):#\n            x=1\n        if(content_type_id==0):#   \n            user_lecture_sum[i] = user_lecture_sum_dict[user_id]\n            user_lecture_count[i] = user_lecture_count_dict[user_id]\n                \n            user_sum[i] = user_sum_dict[user_id]\n            user_count[i] = user_count_dict[user_id]\n            task_container_sum[i] = task_container_sum_dict[task_container_id]\n            task_container_count[i] = task_container_count_dict[task_container_id]\n            task_container_std[i]=task_container_std_dict[task_container_id]\n\n            explanation_sum_dict[user_id] += prior_question_had_explanation\n            explanation_count_dict[user_id] += 1\n            explanation_sum[i] = explanation_sum_dict[user_id]\n            explanation_count[i] = explanation_count_dict[user_id]\n\n            if user_id in max_timestamp_u_dict['max_time_stamp'].keys():\n                lagtime[i]=timestamp-max_timestamp_u_dict['max_time_stamp'][user_id]\n                if(max_timestamp_u_dict2['max_time_stamp2'][user_id]==lagtime_mean2):#\n                    lagtime2[i]=lagtime_mean2\n                    lagtime3[i]=lagtime_mean3\n                    #max_timestamp_u_dict3['max_time_stamp3'].update({user_id:lagtime_mean3})\n                else:\n                    lagtime2[i]=timestamp-max_timestamp_u_dict2['max_time_stamp2'][user_id]\n                    if(max_timestamp_u_dict3['max_time_stamp3'][user_id]==lagtime_mean3):\n                        lagtime3[i]=lagtime_mean3 #\n                    else:\n                        lagtime3[i]=timestamp-max_timestamp_u_dict3['max_time_stamp3'][user_id]\n                    \n                    max_timestamp_u_dict3['max_time_stamp3'][user_id]=max_timestamp_u_dict2['max_time_stamp2'][user_id]\n                        \n                max_timestamp_u_dict2['max_time_stamp2'][user_id]=max_timestamp_u_dict['max_time_stamp'][user_id]\n                max_timestamp_u_dict['max_time_stamp'][user_id]=timestamp\n            else:\n                lagtime[i]=lagtime_mean\n                max_timestamp_u_dict['max_time_stamp'].update({user_id:timestamp})\n                lagtime2[i]=lagtime_mean2#\n                max_timestamp_u_dict2['max_time_stamp2'].update({user_id:lagtime_mean2})\n                lagtime3[i]=lagtime_mean3#\n                max_timestamp_u_dict3['max_time_stamp3'].update({user_id:lagtime_mean3})\n            if user_id in user_prior_question_elapsed_time_dict['prior_question_elapsed_time'].keys():            \n                delta_prior_question_elapsed_time[i]=prior_question_elapsed_time-user_prior_question_elapsed_time_dict['prior_question_elapsed_time'][user_id]\n                user_prior_question_elapsed_time_dict['prior_question_elapsed_time'][user_id]=prior_question_elapsed_time\n            else:           \n                delta_prior_question_elapsed_time[i]=delta_prior_question_elapsed_time_mean    \n                user_prior_question_elapsed_time_dict['prior_question_elapsed_time'].update({user_id:prior_question_elapsed_time})\n            i=i+1 \n\n\n        \n    test_df = test_df[test_df['content_type_id'] == 0].reset_index(drop=True)\n    test_df = test_df.merge(questions_df.loc[questions_df.index.isin(test_df['content_id'])],\n                  how='left', on='content_id', right_index=True) \n    test_df['user_lecture_lv'] = user_lecture_sum / user_lecture_count\n    test_df['user_lecture_sum'] = user_lecture_sum\n    \n    test_df['user_interaction_count'] = user_lecture_count\n    test_df['user_interaction_timestamp_mean'] = test_df['timestamp']/user_lecture_count\n    \n    test_df['user_correctness'] = user_sum / user_count\n    test_df['user_uncorrect_count'] =user_count-user_sum\n    test_df['user_correct_count'] =user_sum\n    test_df['task_container_correctness'] = task_container_sum / task_container_count\n    test_df['task_container_cor_count'] = task_container_sum \n    test_df['task_container_uncor_count'] =task_container_count-task_container_sum \n    test_df['task_container_std'] = task_container_std \n    \n    test_df['explanation_mean'] = explanation_sum / explanation_count\n    test_df['explanation_true_count'] = explanation_sum\n    test_df['explanation_false_count'] = explanation_count-explanation_sum \n    test_df['delta_prior_question_elapsed_time'] = delta_prior_question_elapsed_time \n    test_df[\"attempt_no\"] = test_df[[\"user_id\", \"content_id\"]].apply(lambda row: get_max_attempt(row[\"user_id\"], row[\"content_id\"]), axis=1)\n    test_df[\"lagtime\"]=lagtime\n    test_df[\"lagtime2\"]=lagtime2\n    test_df[\"lagtime3\"]=lagtime3\n    test_df['timestamp']=test_df['timestamp']/(1000*3600)\n    test_df.timestamp=test_df.timestamp.astype('float16')\n    test_df['lagtime']=test_df['lagtime']/(1000*3600)\n    test_df.lagtime=test_df.lagtime.astype('float32')\n    test_df['lagtime2']=test_df['lagtime2']/(1000*3600)\n    test_df.lagtime2=test_df.lagtime2.astype('float32')\n    test_df['lagtime3']=test_df['lagtime3']/(1000*3600)\n    test_df.lagtime3=test_df.lagtime3.astype('float32')\n    test_df['user_interaction_timestamp_mean']=test_df['user_interaction_timestamp_mean']/(1000*3600)\n    test_df.user_interaction_timestamp_mean=test_df.user_interaction_timestamp_mean.astype('float32')\n    \n    test_df['user_correctness'].fillna(0.67, inplace=True)\n\n    sub_preds = np.zeros(test_df.shape[0])\n    for i, model in enumerate(clfs, 1):\n        test_preds  = model.predict(test_df[features])\n        sub_preds += test_preds\n    o2=sub_preds / len(clfs)\n    \n    test_df[target] = 0.5 *np.array(outs) + 0.5 *np.array(o2)\n    env.predict(test_df[['row_id', target]])","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.097159,"end_time":"2021-01-01T15:33:08.982017","exception":false,"start_time":"2021-01-01T15:33:08.884858","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}