{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"!pip install ../input/python-datatable/datatable-0.11.0-cp37-cp37m-manylinux2010_x86_64.whl > /dev/null 2>&1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom collections import defaultdict\nimport datatable as dt\nimport lightgbm as lgb\nfrom matplotlib import pyplot as plt\nimport riiideducation\n\n_ = np.seterr(divide='ignore', invalid='ignore')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Preprocess"},{"metadata":{"trusted":true},"cell_type":"code","source":"data_types_dict = {\n    'user_id': 'int32', \n    'content_id': 'int16', \n    'task_container_id': 'int16',\n    'answered_correctly': 'int8', \n    'prior_question_elapsed_time': 'float32', \n    'prior_question_had_explanation': 'bool'\n}\ntarget = 'answered_correctly'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"example_test = pd.read_csv('../input/riiid-test-answer-prediction/example_test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = dt.fread('../input/riiid-test-answer-prediction/train.csv', columns=set(data_types_dict.keys())).to_pandas()\ntrain_df = train_df[train_df[target] != -1].reset_index(drop=True)\ntrain_df['prior_question_had_explanation'].fillna(False, inplace=True)\ntrain_df = train_df.astype(data_types_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### target"},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_target_by_user = train_df.groupby('user_id')[target].agg(['sum', 'count', \"std\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_target_by_content = train_df.groupby('content_id')[target].agg(['sum', 'count', \"std\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_target_by_user_content = train_df.groupby([\"user_id\",'content_id'])[target].agg(['sum', 'count', \"std\"])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### prior_question_elapsed_time"},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_prior_question_elapsed_time_by_user = train_df.groupby('user_id')[\"prior_question_elapsed_time\"].agg(['min', \"max\",\"mean\", \"std\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df.groupby('user_id').tail(24).reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## questions_df"},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df = pd.read_csv(\n    '../input/riiid-test-answer-prediction/questions.csv', \n    usecols=[0, 3],\n    dtype={'question_id': 'int16', 'part': 'int8'}\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def feature_engineering(_df):\n    _df = pd.merge(_df, questions_df, left_on='content_id', right_on='question_id', how='left')\n    _df.drop(columns=['question_id'], inplace=True)\n    _df['user_target_mean']  = _df['user_id'].map(agg_target_by_user['sum'] / agg_target_by_user['count'])\n    _df['content_target_count'] = _df['content_id'].map(agg_target_by_content['count']).astype('int32')\n    _df['content_target_mean'] = _df['content_id'].map(agg_target_by_content['sum'] / agg_target_by_content['count'])\n    \n    return _df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = feature_engineering(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_df = train_df.groupby('user_id').tail(6)\ntrain_df.drop(valid_df.index, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train"},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {\n    'objective': 'binary',\n    'seed': 42,\n    'metric': 'auc',\n    'learning_rate': 0.05,\n    'max_bin': 800,\n    'num_leaves': 80\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features = set(train_df.columns) - {\"answered_correctly\", \"user_id\", \"content_id\", \"task_container_id\"}\nfeatures = list(features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tr_data = lgb.Dataset(train_df[features], label=train_df[target])\nva_data = lgb.Dataset(valid_df[features], label=valid_df[target])\n\nmodel = lgb.train(\n    params, \n    tr_data, \n    num_boost_round=10000,\n    valid_sets=[tr_data, va_data], \n    early_stopping_rounds=50,\n    verbose_eval=50\n)\n\n# model.save_model(f'model.txt')\nlgb.plot_importance(model, importance_type='gain')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## predict"},{"metadata":{"trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()\niter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    test_df['prior_question_had_explanation'] = test_df['prior_question_had_explanation'].fillna(False).astype('bool')\n    test_df = test_df[test_df['content_type_id'] == 0].reset_index(drop=True)\n    test_df = feature_engineering(test_df)\n    test_df[target] = model.predict(test_df[features])\n    env.predict(test_df[['row_id', target]])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}