{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\n\nimport riiideducation\nfrom sklearn.metrics import roc_auc_score\n\nfrom sklearn.feature_selection import RFE\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import KFold\n\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.linear_model import LogisticRegression\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\n    '/kaggle/input/riiid-test-answer-prediction/train.csv',\n    usecols=[1, 2, 3, 4, 5, 7, 8, 9],\n    dtype={\n        'timestamp': 'int64',\n        'user_id': 'int32',\n        'content_id': 'int16',\n        'content_type_id': 'int8',\n        'task_container_id': 'int16',\n        'answered_correctly':'int8',\n        'prior_question_elapsed_time': 'float32',\n        'prior_question_had_explanation': 'boolean'\n    }\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df = pd.read_csv(\n    '/kaggle/input/riiid-test-answer-prediction/questions.csv',\n    usecols=[0,3],\n    dtype = {\n        \"question_id\":\"int64\",\n        \"part\":\"int8\"\n    }\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures_df = pd.read_csv(\n    '/kaggle/input/riiid-test-answer-prediction/lectures.csv',\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures_df[\"type_of\"] = lectures_df[\"type_of\"].replace(\"solving question\", \"solving_question\")\nlectures_df = pd.get_dummies(lectures_df, columns=[\"part\", \"type_of\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"part_lectures_columns = [column for column in lectures_df.columns if column.startswith(\"part\")]\ntype_of_lectures_columns = [column for column in lectures_df.columns if column.startswith(\"type_of_\")]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_lectures = train[train.content_type_id==True].merge(lectures_df,\n                                                          left_on=\"content_id\",\n                                                          right_on=\"lecture_id\",\n                                                          how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_lecture_stats_part = train_lectures.groupby(\"user_id\")[part_lectures_columns + type_of_lectures_columns].sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_lecture_stats_part.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for column in user_lecture_stats_part.columns:\n    bool_column = column + \"_boolean\"\n    user_lecture_stats_part[bool_column] = (user_lecture_stats_part[column] > 0).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_lecture_stats_part.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del train_lectures\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train.content_type_id == False].sort_values(\"timestamp\").reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"elapsed_mean = train.prior_question_elapsed_time.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#average numbers of seeing questions per user\ngroup1 = train[[\"task_container_id\", \"user_id\"]].groupby(\"task_container_id\").agg([\"count\"])\ngroup1.columns = [\"avg_questions\"]\ngroup2 = train[[\"task_container_id\", \"user_id\"]].groupby(\"task_container_id\").agg([\"nunique\"])\ngroup2.columns = [\"avg_questions\"]\n\ngroup3 = group1 / group2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"group3[\"avg_question_seen\"] = group3.avg_questions.cumsum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_u_final = train[[\"user_id\", \"answered_correctly\"]].groupby(\"user_id\").agg([\"mean\"])\nresults_u_final.columns = [\"answered_correctly_user\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_u2_final = train[[\"user_id\", \"prior_question_had_explanation\"]].groupby(\"user_id\").agg([\"mean\"])\nresults_u2_final.columns = [\"explanation_mean_user\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior_mean_user = results_u2_final.explanation_mean_user.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.merge(train, questions_df, left_on=\"content_id\", right_on=\"question_id\", how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_q_final = train[['question_id','answered_correctly']].groupby(['question_id']).agg(['mean'])\nresults_q_final.columns = ['quest_pct']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_q2_final = train[['question_id','part']].groupby(['question_id']).agg(['count'])\nresults_q2_final.columns = ['quest_count']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question2 = pd.merge(questions_df, results_q2_final, on=\"question_id\", how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question2 = pd.merge(question2, results_q_final, on=\"question_id\", how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question2.quest_pct = round(question2.quest_pct, 5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.drop([\"timestamp\", \"content_type_id\", \"question_id\", \"part\"], axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for validation, extract the five recent data of users\nvalidation = train.groupby(\"user_id\").tail(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[~train.index.isin(validation.index)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_u_val = train[[\"user_id\", \"answered_correctly\"]].groupby(\"user_id\").agg([\"mean\"])\nresults_u_val.columns = [\"answered_correctly_user\"]\n\nresults_u2_val = train[[\"user_id\", \"prior_question_had_explanation\"]].groupby(\"user_id\").agg([\"mean\"])\nresults_u2_val.columns = [\"explanation_mean_user\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train.groupby(\"user_id\").tail(18)\ntrain = train[~train.index.isin(X.index)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train) + len(X) + len(validation)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results_u_X = train[[\"user_id\", \"answered_correctly\"]].groupby(\"user_id\").agg([\"mean\"])\nresults_u_X.columns = [\"answered_correctly_user\"]\n\nresults_u2_X = train[[\"user_id\", \"prior_question_had_explanation\"]].groupby(\"user_id\").agg([\"mean\"])\nresults_u2_X.columns = [\"explanation_mean_user\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del(train)\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = pd.merge(X, group3, left_on=\"task_container_id\", right_index=True, how=\"left\")\nX = pd.merge(X, results_u_X, on=\"user_id\", how=\"left\")\nX = pd.merge(X, results_u2_X, on=\"user_id\", how=\"left\")\n\nX = pd.merge(X, user_lecture_stats_part, on=\"user_id\", how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"validation = pd.merge(validation, group3, left_on=\"task_container_id\", right_index=True, how=\"left\")\nvalidation = pd.merge(validation, results_u_val, on=\"user_id\", how=\"left\")\nvalidation = pd.merge(validation, results_u2_val, on=\"user_id\", how=\"left\")\n\nvalidation = pd.merge(validation, user_lecture_stats_part, on=\"user_id\", how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlb_make = LabelEncoder()\n\nX.prior_question_had_explanation.fillna(False, inplace=True)\nvalidation.prior_question_had_explanation.fillna(False, inplace=True)\n\n#False->0, True->1\nvalidation[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(validation[\"prior_question_had_explanation\"])\nX[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(X[\"prior_question_had_explanation\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"content_mean = question2.quest_pct.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question2.quest_pct = question2.quest_pct.mask((question2[\"quest_count\"] < 3), .65)\n\nquestion2.quest_pct = question2.quest_pct.mask((question2[\"quest_pct\"] < .2) & (question2[\"quest_count\"] < 21), .2)\n\nquestion2.quest_pct = question2.quest_pct.mask((question2[\"quest_pct\"] > .95) & (question2[\"quest_count\"] < 21), .95)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = pd.merge(X, question2, left_on=\"content_id\", right_on=\"question_id\", how=\"left\")\nvalidation = pd.merge(validation, question2, left_on=\"content_id\", right_on=\"question_id\", how=\"left\")\n\nX.part = X.part -1\nvalidation.part = validation.part -1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = X[\"answered_correctly\"]\nX = X.drop([\"answered_correctly\"], axis=1)\n\ny_val = validation[\"answered_correctly\"]\nX_val = validation.drop([\"answered_correctly\"], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = X[['answered_correctly_user', 'explanation_mean_user', 'quest_pct', 'avg_question_seen',\n       'prior_question_elapsed_time','prior_question_had_explanation_enc', 'part',\n       'part_1', 'part_2', 'part_3', 'part_4', 'part_5', 'part_6', 'part_7',\n       'type_of_concept', 'type_of_intention', 'type_of_solving_question', 'type_of_starter',\n       'part_1_boolean', 'part_2_boolean', 'part_3_boolean', 'part_4_boolean', 'part_5_boolean', 'part_6_boolean', 'part_7_boolean',\n       'type_of_concept_boolean', 'type_of_intention_boolean', 'type_of_solving_question_boolean', 'type_of_starter_boolean']]\n\nX_val = X_val[['answered_correctly_user', 'explanation_mean_user', 'quest_pct', 'avg_question_seen',\n               'prior_question_elapsed_time','prior_question_had_explanation_enc', 'part',\n               'part_1', 'part_2', 'part_3', 'part_4', 'part_5', 'part_6', 'part_7',\n               'type_of_concept', 'type_of_intention', 'type_of_solving_question', 'type_of_starter',\n               'part_1_boolean', 'part_2_boolean', 'part_3_boolean', 'part_4_boolean', 'part_5_boolean', 'part_6_boolean', 'part_7_boolean',\n               'type_of_concept_boolean', 'type_of_intention_boolean', 'type_of_solving_question_boolean', 'type_of_starter_boolean']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X['answered_correctly_user'].fillna(0.65,  inplace=True)\nX['explanation_mean_user'].fillna(prior_mean_user,  inplace=True)\nX['quest_pct'].fillna(content_mean, inplace=True)\n\nX['part'].fillna(4, inplace = True)\nX['avg_question_seen'].fillna(1, inplace = True)\nX['prior_question_elapsed_time'].fillna(elapsed_mean, inplace = True)\nX['prior_question_had_explanation_enc'].fillna(0, inplace = True)\n\nX['part_1'].fillna(0, inplace = True)\nX['part_2'].fillna(0, inplace = True)\nX['part_3'].fillna(0, inplace = True)\nX['part_4'].fillna(0, inplace = True)\nX['part_5'].fillna(0, inplace = True)\nX['part_6'].fillna(0, inplace = True)\nX['part_7'].fillna(0, inplace = True)\nX['type_of_concept'].fillna(0, inplace = True)\nX['type_of_intention'].fillna(0, inplace = True)\nX['type_of_solving_question'].fillna(0, inplace = True)\nX['type_of_starter'].fillna(0, inplace = True)\nX['part_1_boolean'].fillna(0, inplace = True)\nX['part_2_boolean'].fillna(0, inplace = True)\nX['part_3_boolean'].fillna(0, inplace = True)\nX['part_4_boolean'].fillna(0, inplace = True)\nX['part_5_boolean'].fillna(0, inplace = True)\nX['part_6_boolean'].fillna(0, inplace = True)\nX['part_7_boolean'].fillna(0, inplace = True)\nX['type_of_concept_boolean'].fillna(0, inplace = True)\nX['type_of_intention_boolean'].fillna(0, inplace = True)\nX['type_of_solving_question_boolean'].fillna(0, inplace = True)\nX['type_of_starter_boolean'].fillna(0, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_val['answered_correctly_user'].fillna(0.65,  inplace=True)\nX_val['explanation_mean_user'].fillna(prior_mean_user,  inplace=True)\nX_val['quest_pct'].fillna(content_mean,  inplace=True)\n\nX_val['part'].fillna(4, inplace = True)\nX_val['avg_question_seen'].fillna(1, inplace = True)\nX_val['prior_question_elapsed_time'].fillna(elapsed_mean, inplace = True)\nX_val['prior_question_had_explanation_enc'].fillna(0, inplace = True)\n\nX_val['part_1'].fillna(0, inplace = True)\nX_val['part_2'].fillna(0, inplace = True)\nX_val['part_3'].fillna(0, inplace = True)\nX_val['part_4'].fillna(0, inplace = True)\nX_val['part_5'].fillna(0, inplace = True)\nX_val['part_6'].fillna(0, inplace = True)\nX_val['part_7'].fillna(0, inplace = True)\nX_val['type_of_concept'].fillna(0, inplace = True)\nX_val['type_of_intention'].fillna(0, inplace = True)\nX_val['type_of_solving_question'].fillna(0, inplace = True)\nX_val['type_of_starter'].fillna(0, inplace = True)\nX_val['part_1_boolean'].fillna(0, inplace = True)\nX_val['part_2_boolean'].fillna(0, inplace = True)\nX_val['part_3_boolean'].fillna(0, inplace = True)\nX_val['part_4_boolean'].fillna(0, inplace = True)\nX_val['part_5_boolean'].fillna(0, inplace = True)\nX_val['part_6_boolean'].fillna(0, inplace = True)\nX_val['part_7_boolean'].fillna(0, inplace = True)\nX_val['type_of_concept_boolean'].fillna(0, inplace = True)\nX_val['type_of_intention_boolean'].fillna(0, inplace = True)\nX_val['type_of_solving_question_boolean'].fillna(0, inplace = True)\nX_val['type_of_starter_boolean'].fillna(0, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {\n    'num_leaves': 31, \n    'n_estimators': 200, \n    'max_depth': 8, \n    'min_child_samples': 356, \n    'learning_rate': 0.2982483634778906, \n    'min_data_in_leaf': 82, \n    'bagging_fraction': 0.6545628633239445, \n    'feature_fraction': 0.9164482379289846,\n    'random_state': 666\n}\n\nfull_model = LGBMClassifier(**params)\nfull_model.fit(X, y)\n\npreds = full_model.predict_proba(X_val)[:,1]\nprint(\"LGB roc auc\", roc_auc_score(y_val, preds))\n\nfull_xgb = XGBClassifier(random_state=666)\nfull_xgb.fit(X, y)\n\npreds = full_xgb.predict_proba(X_val)[:,1]\nprint(\"XGB roc auc\", roc_auc_score(y_val, preds))\n\nfull_lr = LogisticRegression(random_state=666)\nfull_lr.fit(X, y)\n\npreds = full_lr.predict_proba(X_val)[:,1]\nprint(\"LR roc auc\", roc_auc_score(y_val, preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import optuna\nfrom optuna.samplers import TPESampler","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rfe = RFE(estimator=DecisionTreeClassifier(random_state=666), n_features_to_select=14)\nrfe.fit(X, y)\nX = rfe.transform(X)\nX_val = rfe.transform(X_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sampler = TPESampler(seed=666)\n\n# def create_model(trial):\n#     num_leaves = trial.suggest_int(\"num_leaves\", 2, 31)\n#     n_estimators = trial.suggest_int(\"n_estimators\", 20, 300)\n#     max_depth = trial.suggest_int('max_depth', 3, 9)\n#     min_child_samples = trial.suggest_int('min_child_samples', 100, 1200)\n#     learning_rate = trial.suggest_uniform('learning_rate', 0.0001, 0.99)\n#     min_data_in_leaf = trial.suggest_int('min_data_in_leaf', 5, 90)\n#     bagging_fraction = trial.suggest_uniform('bagging_fraction', 0.0001, 1.0)\n#     feature_fraction = trial.suggest_uniform('feature_fraction', 0.0001, 1.0)\n#     model = LGBMClassifier(\n#         num_leaves=num_leaves,\n#         n_estimators=n_estimators, \n#         max_depth=max_depth, \n#         min_child_samples=min_child_samples, \n#         min_data_in_leaf=min_data_in_leaf,\n#         learning_rate=learning_rate,\n#         feature_fraction=feature_fraction,\n#         random_state=666\n# )\n#     return model\n\n# def objective(trial):\n#     model = create_model(trial)\n#     model.fit(X, y)\n#     preds = model.predict_proba(X_val)[:,1]\n#     score = roc_auc_score(y_val, preds)\n#     return score\n\n# # run optuna \n# study = optuna.create_study(direction=\"maximize\", sampler=sampler)\n# study.optimize(objective, n_trials=350)\n# params = study.best_params\n# params['random_state'] = 666\n\n#  ↑ After Trial=286 ended, 9hours run-time-limit was reached.\n\n# Referring to the previous attempt, narrow down the range of hyperparameters\ndef create_model(trial):\n    num_leaves = trial.suggest_int(\"num_leaves\", 26, 32)\n    n_estimators = trial.suggest_int(\"n_estimators\", 280, 350)\n    max_depth = trial.suggest_int('max_depth', 7, 9)\n    min_child_samples = trial.suggest_int('min_child_samples', 1000, 1200)\n    learning_rate = trial.suggest_uniform('learning_rate', 0.1, 0.5)\n    min_data_in_leaf = trial.suggest_int('min_data_in_leaf', 25, 90)\n    bagging_fraction = trial.suggest_uniform('bagging_fraction', 0.1, 1.0)\n    feature_fraction = trial.suggest_uniform('feature_fraction', 0.1, 1.0)\n    model = LGBMClassifier(\n        num_leaves=num_leaves,\n        n_estimators=n_estimators, \n        max_depth=max_depth, \n        min_child_samples=min_child_samples, \n        min_data_in_leaf=min_data_in_leaf,\n        learning_rate=learning_rate,\n        feature_fraction=feature_fraction,\n        random_state=666\n)\n    return model\n\ndef objective(trial):\n    model = create_model(trial)\n    model.fit(X, y)\n    preds = model.predict_proba(X_val)[:,1]\n    score = roc_auc_score(y_val, preds)\n    return score\n\n# run optuna \n# study = optuna.create_study(direction=\"maximize\", sampler=sampler)\n# study.optimize(objective, n_trials=200)\n# params = study.best_params\n# params['random_state'] = 666","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {'num_leaves': 29,\n 'n_estimators': 300,\n 'max_depth': 9,\n 'min_child_samples': 1089,\n 'learning_rate': 0.306030368670154,\n 'min_data_in_leaf': 65,\n 'bagging_fraction': 0.49498535405259425,\n 'feature_fraction': 0.9235503880887722,\n 'random_state': 666}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = LGBMClassifier(**params)\nmodel.fit(X, y)\n\npreds = model.predict_proba(X_val)[:,1]\nroc_auc_score(y_val, preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = pd.DataFrame(X)\nX_val = pd.DataFrame(X_val)\n\ny = pd.DataFrame(y)\ny_val = pd.DataFrame(y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"models = []\npreds = []\nfor n, (tr, te) in enumerate(KFold(n_splits=5, random_state=666, shuffle=True).split(y)):\n    print(f'Fold {n}')\n    model = LGBMClassifier(**params)\n    model.fit(X.values[tr], y.values[tr])\n    \n    pred = model.predict_proba(X_val)[:, 1]\n    preds.append(pred)\n    print('Fold roc auc:', roc_auc_score(y.values[te], model.predict_proba(X.values[te])[:, 1])) \n    models.append(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = preds[0]\nfor i in range(1,5):\n    predictions += preds[i]\npredictions /= 5\n\nprint(\"ROC AUC\", roc_auc_score(y_val, predictions))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"iter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    test_df['task_container_id'] = test_df.task_container_id.mask(test_df.task_container_id > 9999, 9999)\n    test_df = pd.merge(test_df, group3, left_on=['task_container_id'], right_index= True, how=\"left\")\n    test_df = pd.merge(test_df, question2, left_on = 'content_id', right_on = 'question_id', how = 'left')\n    test_df = pd.merge(test_df, results_u_final, on=['user_id'],  how=\"left\")\n    test_df = pd.merge(test_df, results_u2_final, on=['user_id'],  how=\"left\")\n    \n    test_df = pd.merge(test_df, user_lecture_stats_part, on=['user_id'], how=\"left\")\n    test_df['part_1'].fillna(0, inplace = True)\n    test_df['part_2'].fillna(0, inplace = True)\n    test_df['part_3'].fillna(0, inplace = True)\n    test_df['part_4'].fillna(0, inplace = True)\n    test_df['part_5'].fillna(0, inplace = True)\n    test_df['part_6'].fillna(0, inplace = True)\n    test_df['part_7'].fillna(0, inplace = True)\n    test_df['type_of_concept'].fillna(0, inplace = True)\n    test_df['type_of_intention'].fillna(0, inplace = True)\n    test_df['type_of_solving_question'].fillna(0, inplace = True)\n    test_df['type_of_starter'].fillna(0, inplace = True)\n    test_df['part_1_boolean'].fillna(0, inplace = True)\n    test_df['part_2_boolean'].fillna(0, inplace = True)\n    test_df['part_3_boolean'].fillna(0, inplace = True)\n    test_df['part_4_boolean'].fillna(0, inplace = True)\n    test_df['part_5_boolean'].fillna(0, inplace = True)\n    test_df['part_6_boolean'].fillna(0, inplace = True)\n    test_df['part_7_boolean'].fillna(0, inplace = True)\n    test_df['type_of_concept_boolean'].fillna(0, inplace = True)\n    test_df['type_of_intention_boolean'].fillna(0, inplace = True)\n    test_df['type_of_solving_question_boolean'].fillna(0, inplace = True)\n    test_df['type_of_starter_boolean'].fillna(0, inplace = True)\n    \n    test_df['answered_correctly_user'].fillna(0.65,  inplace=True)\n    test_df['explanation_mean_user'].fillna(prior_mean_user,  inplace=True)\n    test_df['quest_pct'].fillna(content_mean,  inplace=True)\n    test_df['part'] = test_df.part - 1\n\n    test_df['part'].fillna(4, inplace = True)\n    test_df['avg_question_seen'].fillna(1, inplace = True)\n    test_df['prior_question_elapsed_time'].fillna(elapsed_mean, inplace = True)\n    test_df['prior_question_had_explanation'].fillna(False, inplace=True)\n    test_df[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(test_df[\"prior_question_had_explanation\"])\n    \n    full_preds = full_model.predict_proba(test_df[['answered_correctly_user', 'explanation_mean_user', 'quest_pct', 'avg_question_seen',\n                                                            'prior_question_elapsed_time','prior_question_had_explanation_enc', 'part',\n                                                            'part_1', 'part_2', 'part_3', 'part_4', 'part_5', 'part_6', 'part_7',\n                                                            'type_of_concept', 'type_of_intention', 'type_of_solving_question', 'type_of_starter',\n                                                            'part_1_boolean', 'part_2_boolean', 'part_3_boolean', 'part_4_boolean', 'part_5_boolean', 'part_6_boolean', 'part_7_boolean',\n                                                            'type_of_concept_boolean', 'type_of_intention_boolean', 'type_of_solving_question_boolean', 'type_of_starter_boolean']])[:, 1]\n    \n    full_preds_xgb = full_xgb.predict_proba(test_df[['answered_correctly_user', 'explanation_mean_user', 'quest_pct', 'avg_question_seen',\n                                                            'prior_question_elapsed_time','prior_question_had_explanation_enc', 'part',\n                                                            'part_1', 'part_2', 'part_3', 'part_4', 'part_5', 'part_6', 'part_7',\n                                                            'type_of_concept', 'type_of_intention', 'type_of_solving_question', 'type_of_starter',\n                                                            'part_1_boolean', 'part_2_boolean', 'part_3_boolean', 'part_4_boolean', 'part_5_boolean', 'part_6_boolean', 'part_7_boolean',\n                                                            'type_of_concept_boolean', 'type_of_intention_boolean', 'type_of_solving_question_boolean', 'type_of_starter_boolean']])[:, 1]\n    \n    full_preds_lr = full_lr.predict_proba(test_df[['answered_correctly_user', 'explanation_mean_user', 'quest_pct', 'avg_question_seen',\n                                                            'prior_question_elapsed_time','prior_question_had_explanation_enc', 'part',\n                                                            'part_1', 'part_2', 'part_3', 'part_4', 'part_5', 'part_6', 'part_7',\n                                                            'type_of_concept', 'type_of_intention', 'type_of_solving_question', 'type_of_starter',\n                                                            'part_1_boolean', 'part_2_boolean', 'part_3_boolean', 'part_4_boolean', 'part_5_boolean', 'part_6_boolean', 'part_7_boolean',\n                                                            'type_of_concept_boolean', 'type_of_intention_boolean', 'type_of_solving_question_boolean', 'type_of_starter_boolean']])[:, 1]\n    \n\n\n    \n    X_test = rfe.transform(test_df[['answered_correctly_user', 'explanation_mean_user', 'quest_pct', 'avg_question_seen',\n                                                            'prior_question_elapsed_time','prior_question_had_explanation_enc', 'part',\n                                                            'part_1', 'part_2', 'part_3', 'part_4', 'part_5', 'part_6', 'part_7',\n                                                            'type_of_concept', 'type_of_intention', 'type_of_solving_question', 'type_of_starter',\n                                                            'part_1_boolean', 'part_2_boolean', 'part_3_boolean', 'part_4_boolean', 'part_5_boolean', 'part_6_boolean', 'part_7_boolean',\n                                                            'type_of_concept_boolean', 'type_of_intention_boolean', 'type_of_solving_question_boolean', 'type_of_starter_boolean']])\n    \n    preds = [model.predict_proba(X_test)[:,1] for model in models]\n    \n    predictions = preds[0]\n    for i in range(1, 5):\n        predictions += preds[i]\n    predictions /= 5\n    \n    test_df['answered_correctly'] =  predictions * 0.75 + full_preds * 0.125 + full_preds_xgb * 0.75 + full_preds_lr * 0.05\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}