{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport optuna\nfrom random import sample\nfrom sklearn.metrics import roc_auc_score\nimport gc\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('display.max_columns', None)\n#显示所有行\npd.set_option('display.max_rows', None)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv',\n                   usecols=[1, 2, 3, 4, 5, 7, 8, 9],\n                   dtype={'timestamp': 'int64',\n                          'user_id': 'int32',\n                          'content_id': 'int16',\n                          'content_type_id': 'int8',\n                          'task_container_id': 'int16',\n                          'answered_correctly':'int8',\n                          'prior_question_elapsed_time': 'float32',\n                          'prior_question_had_explanation': 'boolean'}\n                   )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_list = train['user_id'].unique()\nuser_list = list(user_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(user_list)//15","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_user_list = sample(user_list, len(user_list)//15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train['user_id'].isin(sample_user_list)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.sort_values(['user_id','timestamp'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['prior_question_elapsed_time'] = train['prior_question_elapsed_time'].fillna(0)\ntrain['prior_question_had_explanation'] = train['prior_question_had_explanation'].fillna(False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')\nlectures['content_type_id'] = 1\nlectures = lectures.rename(columns={'lecture_id':'content_id','tag':'lecture_tag','part':'lecture_part'})\nlectures['type_of'] = lectures['type_of'].replace('solving question', 'solving_question')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lecture_tag_count =  lectures['lecture_tag'].value_counts().to_dict()\nlectures['lecture_tag'] = lectures['lecture_tag'].map(lecture_tag_count)\n# lectures['lecture_tag'] = lectures['lecture_tag'].map({3:'most',4:'second',2:'second',6:'third',5:'third',1:'third',7:'third'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures['lecture_tag'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures['lecture_tag'] = lectures['lecture_tag'].apply(lambda x : 6 if ((x==7)|(x==5)) else x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures['lecture_tag'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures = pd.get_dummies(lectures, columns=['lecture_part','lecture_tag','type_of'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures_column = lectures.columns.to_list()\nlectures_column.remove('content_id')\nlectures_column.remove('content_type_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures.to_csv('lectures.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.merge(train,lectures, on = ['content_id','content_type_id'], how='left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[lectures_column] = train[lectures_column].fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lecture_cumsum = train.groupby('user_id')[lectures_column].cumsum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in lectures_column:\n    lecture_cumsum[i] = lecture_cumsum[i].astype('int32')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lecture_cumsum.max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[lectures_column] = lecture_cumsum","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train['content_type_id']==0].reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"answered_cumsum = train.groupby('user_id')['answered_correctly'].cumsum()\nanswered_count = train.groupby('user_id')['answered_correctly'].cumcount()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"answered_count.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['answered_cumsum'] = answered_cumsum\ntrain['answered_count'] = answered_count\ntrain['answered_cumsum'] = train['answered_cumsum'] - train['answered_correctly']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['user_correctly_rate'] = train['answered_cumsum']/train['answered_count']\ntrain['user_correctly_rate'] = train['user_correctly_rate'].mask((train['answered_count'] < 5), .65)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head(100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"task_info = pd.read_csv('../input/avg-questions-seen/task_info.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.merge(train,task_info,on='task_container_id',how = 'left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question_to_tag = pd.read_csv('../input/riiid-question-to-tag/question_to_tag.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.merge(train,question_to_tag,left_on = 'content_id',right_on='question_id',how = 'left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"answered_correctly_content = pd.read_csv('../input/user-content-correctly/answered_correctly_content.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.merge(train,answered_correctly_content,on = 'content_id',how = 'left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['timestamp'] = train['timestamp']//3600000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"edge = sample(sample_user_list, len(sample_user_list)//7)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = train[train['user_id'].isin(edge)]\ntrain = train[~train.index.isin(test.index)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tail = train.groupby('user_id').tail(8)\ntrain = train[~train.index.isin(train_tail.index)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.concat([test,train_tail])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train = train['answered_correctly']\ny_test = test['answered_correctly']\nX_train = train.drop(['user_id','content_id','content_type_id','task_container_id','answered_correctly','question_id'],axis = 1)\nX_test = test.drop(['user_id','content_id','content_type_id','task_container_id','answered_correctly','question_id'],axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# from  sklearn.model_selection import train_test_split\n# X_train,X_test, y_train, y_test =train_test_split(train,target,test_size=0.2, random_state=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train.loc[:,'prior_question_had_explanation']=X_train.loc[:,'prior_question_had_explanation'].astype('bool')\nX_test.loc[:,'prior_question_had_explanation']=X_test.loc[:,'prior_question_had_explanation'].astype('bool')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del train\n# del target\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import roc_auc_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lgb_train = lgb.Dataset(X_train, y_train, free_raw_data=False,\n                        categorical_feature=['prior_question_had_explanation','part','tags1','tags2'])\nlgb_eval = lgb.Dataset(X_test, y_test, free_raw_data=False,\n                        categorical_feature=['prior_question_had_explanation','part','tags1','tags2'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del X_train\ndel y_train\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def objective(trial):    \n    params = {\n            'num_leaves': trial.suggest_int('num_leaves', 100, 500),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n            'max_depth': trial.suggest_int('max_depth', 4, 30),\n            'min_child_weight': trial.suggest_int('min_child_weight', 4, 16),\n            'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0),\n            'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.6, 1.0),\n            'bagging_freq': trial.suggest_int('bagging_freq', 1, 5),\n            'min_child_samples': trial.suggest_int('min_child_samples', 10, 80),\n            'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 1.0),\n            'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 1.0),\n            'is_unbalance':trial.suggest_categorical('is_unbalance', ['-', '+']),\n            'boosting_type': 'gbdt',\n            'objective': 'binary',\n            'metric': 'auc',\n            'early_stopping_rounds': 100\n            }\n\n    model = lgb.train(params, lgb_train, valid_sets=[lgb_train,lgb_eval], verbose_eval=20)\n    val_pred = model.predict(X_test)\n    score = roc_auc_score(y_test, val_pred)\n    print(f\"AUC = {score}\")\n    return score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"study = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number of finished trials: {}'.format(len(study.trials)))\nprint('Best trial:')\ntrial = study.best_trial\nprint('  Value: {}'.format(trial.value))\nprint('  Params: ')\nfor key, value in trial.params.items():\n    print('    {}: {}'.format(key, value))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot history\nfrom optuna.visualization import plot_optimization_history\nplot_optimization_history(study)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = lgb.train(trial.params, lgb_train, valid_sets=[lgb_eval], verbose_eval=1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n#displaying the most important features\nlgb.plot_importance(model)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save_model('model.txt')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}