{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport riiideducation\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/riiid-test-answer-prediction/train.csv', low_memory=False, nrows=10**5,\n                      dtype={\n                          'row_id': 'int64', 'timestamp': 'int64', 'user_id': 'int32', 'content_id': 'int16', 'content_type_id': 'int8',\n                              'task_container_id': 'int16', 'user_answer': 'int8', 'answered_correctly': 'int8', 'prior_question_elapsed_time': 'float32', \n                             'prior_question_had_explanation': 'boolean',\n                      })","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 追加するデータ1 user_answers_df\ntrain_questions_only_df = train_df[train_df['answered_correctly'] != -1]\ngrouped_by_user_df = train_questions_only_df.groupby('user_id')\ngrouped_by_user_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_answers_df = grouped_by_user_df.agg({'answered_correctly': ['mean', 'count']}).copy()\nuser_answers_df.columns = ['mean_user_accuracy', 'questions_answered']\nuser_answers_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 追加するデータ2 questions_df\nquestions_df = pd.read_csv('../input/riiid-test-answer-prediction/questions.csv')\n\ngrouped_by_content_df = train_questions_only_df.groupby('content_id')\n\ncontent_answers_df = grouped_by_content_df.agg({'answered_correctly': ['mean', 'count'] }).copy()\ncontent_answers_df.columns = ['mean_accuracy', 'question_asked']\n\nquestions_df = questions_df.merge(content_answers_df, left_on = 'question_id', right_on = 'content_id', how = 'left')\n\nbundle_dict = questions_df['bundle_id'].value_counts().to_dict()\n\n# right_answers 正解数\nquestions_df['right_answers'] = questions_df['mean_accuracy'] * questions_df['question_asked']\n\nquestions_df['bundle_size'] = questions_df['bundle_id'].apply(lambda x: bundle_dict[x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 追加するデータ3 bundle_answers_df\ngrouped_by_bundle_df = questions_df.groupby('bundle_id')\n\nbundle_answers_df = grouped_by_bundle_df.agg({'right_answers': 'sum', 'question_asked': 'sum'}).copy()\nbundle_answers_df.columns = ['bundle_right_answers', 'bundle_questions_asked']\n\nbundle_answers_df['bundle_accuracy'] = bundle_answers_df['bundle_right_answers'] / bundle_answers_df['bundle_questions_asked']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bundle_answers_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 追加するデータ4 part_answers_df\ngrouped_by_part_df = questions_df.groupby('part')\n\npart_answers_df = grouped_by_part_df.agg({'right_answers': 'sum', 'question_asked': 'sum'}).copy()\n\npart_answers_df.columns = ['part_right_answers', 'part_questions_asked']\npart_answers_df['part_accuracy'] = part_answers_df['part_right_answers'] / part_answers_df['part_questions_asked']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"part_answers_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features = [\n    'timestamp',\n    'mean_user_accuracy', \n    'questions_answered',\n    'mean_accuracy',\n    'question_asked',\n    'prior_question_elapsed_time', \n    'prior_question_had_explanation',\n    'bundle_size', \n    'bundle_accuracy',\n    'part_accuracy', \n    'right_answers',\n    #'user_answer',\n    'correct_answer'\n]\n\ntarget = 'answered_correctly'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 講義(-1)以外を抽出 train\ntrain_part_df = train_df[train_df[target] != -1]\ntrain_part_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# user_answers_df\ntrain_part_df = train_part_df.merge(user_answers_df, how='left', on='user_id')\n\n# questions_df\ntrain_part_df = train_part_df.merge(questions_df, how='left', left_on='content_id', right_on='question_id')\n\n# bundle_answers_df\ntrain_part_df = train_part_df.merge(bundle_answers_df, how='left', on='bundle_id')\n\n# part_answers_df\ntrain_part_df = train_part_df.merge(part_answers_df, how='left', on='part')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ユーザーが質問に回答した後、説明と正しい回答を確認したかどうか 欠損値をFalseと置く、 astypeでデータ型の変換(キャスト)\ntrain_part_df['prior_question_had_explanation'] = train_part_df['prior_question_had_explanation'].fillna(value=False).astype(int)\n\ntrain_part_df.fillna(value = -1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_part_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import lightgbm as lgb\nimport optuna\nfrom sklearn.model_selection import train_test_split,KFold,cross_validate,cross_val_score\nfrom sklearn.metrics import confusion_matrix,accuracy_score, roc_curve, auc\n\ntrain_part_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_part_df = train_part_df.drop('tags',axis=1)\n#X_train = train_part_df.drop(target,axis=1)\nX_train = train_part_df[features] \ny_train = train_part_df[target]\n\n\n# 学習データ = 75% テストデータ = 25%　に分割\ntrain_x, test_x, train_y, test_y = train_test_split(X_train, y_train, test_size=0.25, \n                                                    shuffle = True , random_state = 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_x.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LightGBM用のDatasetに格納\ndtrain = lgb.Dataset(train_x, label=train_y)\ndtest = lgb.Dataset(test_x, label=test_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#------------------------LightGBM Model 最適化-----------------------\n# ハイパーパラメータ検索ライブラリ「Optuna」を使用\n\ndef objectives(trial):\n    \n    #--optunaでのハイパーパラメータサーチ範囲の設定\n    # 二値分類\n    # binary_logloss(クロスエントロピー)最適化\n    # 勾配ブースティングを使用する\n    params = {\n        #fixed\n        'boost_from_average': True, ## ONLY NEED FOR LGB VERSION 2.1.2\n        \"objective\": \"binary\",\n        'boosting_type':'gbdt',\n        'max_depth':-1,\n        'learning_rate':0.1,\n        'n_estimators': 1000,\n        'metric':'binary_logloss',\n\n        #variable\n        'num_leaves': trial.suggest_int('num_leaves', 10, 300),\n        'reg_alpha': trial.suggest_loguniform('reg_alpha',0.001, 10),\n        'reg_lambda':trial.suggest_loguniform('reg_lambda', 0.001, 10),\n    }\n\n    # LightGBMで学習+予測\n    model = lgb.LGBMClassifier(**params,random_state=0)\n    \n    # kFold交差検定で決定係数を算出し、各セットの平均値を返す\n    kf = KFold(n_splits=5, shuffle=True, random_state=0)\n    scores = cross_validate(model, X=train_x, y=train_y,scoring='r2',cv=kf)   \n\n    # 最小化問題とするので1.0から引く\n    return 1.0 - scores['test_score'].mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# optunaによる最適化呼び出し\nopt = optuna.create_study(direction='minimize')\nopt.optimize(objectives, n_trials=20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 実行結果表示\nprint('最終トライアル回数:{}'.format(len(opt.trials)))\nprint('ベストトライアル:')\ntrial = opt.best_trial\nprint('値:{}'.format(trial.value))\nprint('パラメータ:')\nfor key, value in trial.params.items():\n    print('{}:{}'.format(key, value))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error,r2_score,f1_score\n\ngbm_best = lgb.train(trial.params, dtrain)\n\ndef model_Eval_Reg(testX,testY):\n    \n    predict_best = gbm_best.predict(testX)\n    print(predict_best.shape)\n\n    np.set_printoptions(threshold=10)        # 10件表示設定\n    pd.set_option('display.max_rows',10)      # 10件表示設定\n\n    preds = pd.DataFrame({\"preds\":predict_best, \"true\":testY})\n    preds\n\n    # 残差プロット\n    preds[\"residuals\"] = preds[\"true\"] - preds[\"preds\"]\n    preds.plot(x = \"preds\", y = \"residuals\",kind = \"scatter\")\n\n    # モデルのあてはめ\n    fig = plt.figure()\n    ax = fig.add_subplot(1,1,1)\n\n    ax.scatter(preds[\"true\"], preds[\"preds\"],label=\"LigntGBM Model Fitting\")\n    ax.set_xlabel('predicted')\n    ax.set_ylabel('true')\n    ax.set_aspect('equal')\n\n    # rmseとr2を求める\n    rmse = np.sqrt(mean_squared_error(preds[\"true\"], preds[\"preds\"]))\n    print('rmse:',rmse)\n    r2 = r2_score(preds[\"true\"], preds[\"preds\"])\n    print('r2:',r2)\n    \n    # 重要度プロット\n    lgb.plot_importance(gbm_best,importance_type='split',max_num_features = 20,figsize=(12,6))\n    \ndef model_Eval_Cls(testX,testY):\n\n    predict_best = gbm_best.predict(testX)\n    predictions_lgbm = np.where(predict_best > 0.5, 1, 0)\n\n    #lgb.plot_importance(gbm_best,importance_type='split',max_num_features = 20,figsize=(12,6))\n    lgb.plot_importance(gbm_best,importance_type='split',figsize=(12,6))\n\n    # ROC曲線を出力\n    plt.figure()\n    false_positive_rate, recall, thresholds = roc_curve(testY, predict_best)\n    roc_auc = auc(false_positive_rate, recall)\n    plt.title('Receiver Operating Characteristic (ROC)')\n    plt.plot(false_positive_rate, recall, 'b', label = 'AUC = %0.3f' %roc_auc)\n    plt.legend(loc='lower right')\n    plt.plot([0,1], [0,1], 'r--')\n    plt.xlim([0.0,1.0])\n    plt.ylim([0.0,1.0])\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.show()\n    \n    print('AUC Curve score:', roc_auc)\n\n    # 混同行列出力\n    plt.figure()\n    cm = confusion_matrix(testY, predictions_lgbm)\n    labels = ['True', 'False']\n    plt.figure(figsize=(8,6))\n    sns.heatmap(cm, xticklabels = labels, yticklabels = labels, annot = True, fmt='d', cmap=\"Blues\", vmin = 0.2)\n    plt.title('Confusion Matrix')\n    plt.ylabel('True Class')\n    plt.xlabel('Predicted Class')\n    plt.show()\n\n    tp, fn, fp, tn = cm.flatten()\n    print('\\nSpecificity:\\n',(tn / (fp + tn)))\n    print('\\nFalse Negative Rate:\\n',(fn / (tp + fn)))\n    print('\\nFalse Positive Rate:\\n',(fp / (fp + tn)))\n\n    # 正解率プロット\n    print('\\nAccuracy:\\n', accuracy_score(testY, predictions_lgbm))\n    print('\\nAUC:\\n', roc_auc_score(testY, predictions_lgbm))\n    print('\\nConfusion matrix:\\n', confusion_matrix(testY, predictions_lgbm))\n    #print('\\nPrecision:\\n', precision_score(testY, predictions_lgbm))\n    #print('\\nRecall:\\n', recall_score(testY, predictions_lgbm))\n    print('\\nF-measure:\\n', f1_score(testY, predictions_lgbm))\n    \n    print('\\nPrecision:\\n',(tp / (tp + fp)))\n    print('\\nRecall:\\n',(tp / (tp + fn)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# モデルデータの検証\nmodel_Eval_Cls(train_x,train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 過学習の確認\n# テストデータの検証\nmodel_Eval_Cls(test_x,test_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"iter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    y_preds = []\n    \n    test_df = test_df.merge(user_answers_df, how = 'left', on = 'user_id')\n    test_df = test_df.merge(questions_df, how = 'left', left_on = 'content_id', right_on = 'question_id')\n    test_df = test_df.merge(bundle_answers_df, how = 'left', on = 'bundle_id')\n    test_df = test_df.merge(part_answers_df, how = 'left', on = 'part')\n    \n    test_df['prior_question_had_explanation'] = test_df['prior_question_had_explanation'].fillna(value = False).astype(bool)\n    test_df.fillna(value = -1, inplace = True)\n    \n    y_pred = gbm_best.predict(test_df[features])\n    #y_pred = gbm_best.predict(X_train)\n    pred_lgbm = np.where(y_pred > 0.5, 1, 0)\n    y_preds.append(pred_lgbm)\n    \n    #for mdl in gbm_best:\n    #    y_pred = mdl.predict(test_df[features], num_iteration=mdl.best_iteration)\n    #    y_preds.append(y_pred)\n        \n    y_preds = sum(y_preds) / len(y_preds)\n    test_df['answered_correctly'] = y_preds\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}