{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport os\nfrom typing import List, Dict, Optional\nimport numpy as np\nfrom sklearn.model_selection import RepeatedKFold\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport math\nimport time\nimport random\nimport lightgbm as lgb\nimport gc\nimport os\nfrom collections import defaultdict\nimport datatable as dt\nfrom sklearn.preprocessing import LabelEncoder\nfrom numba import jit\nfrom sklearn.model_selection import StratifiedKFold, KFold, RepeatedKFold, GroupKFold, GridSearchCV, train_test_split, TimeSeriesSplit\nfrom sklearn import metrics\nimport riiideducation\n\n_ = np.seterr(divide='ignore', invalid='ignore')\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"train.csv\n\n1. **row_id**: (int64) ID code for the row.\n\n2. **timestamp**: (int64) the time between this user interaction and the first event from that user.\n\n    user_id: (int32) ID code for the user.\n\n3. **content_id**: (int16) ID code for the user interaction\n\n4. **content_type_id**: (int8) 0 if the event was a question being posed to the user, 1 if the event was \n    the user watching a lecture.\n\n5. **task_container_id**: (int16) Id code for the batch of questions or lectures. For example, a user might \n    see three questions in a row before seeing the explanations for any of them. Those three would\n    all share a task_container_id. Monotonically increasing for each user.\n\n6. **user_answer**: (int8) the user's answer to the question, if any. Read -1 as null, for lectures.\n\n7. **answered_correctly**: (int8) if the user responded correctly. Read -1 as null, for lectures.\n\n8. **prior_question_elapsed_time**: (float32) How long it took a user to answer their previous question      bundle, ignoring any lectures in between. The value is shared across a single question bundle, and is   null for a user's first question bundle or lecture. Note that the time is the total time a user took to \n   solve all the questions in the previous bundle.\n\n9. **prior_question_had_explanation**: (bool) Whether or not the user saw an explanation and the correct \n    response(s) after answering the previous question bundle, ignoring any lectures in between. \n    The value is shared across a single question bundle, and is null for a user's first question \n    bundle or lecture. Typically the first several questions a user sees were part of an onboarding\n    diagnostic test where they did not get any feedback.\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"data_types_dict = {\n    'timestamp': 'int64',\n    'user_id': 'int32', \n    'content_id': 'int16', \n    'content_type_id':'int8', \n    'task_container_id': 'int16',\n    #'user_answer': 'int8',\n    'answered_correctly': 'int8', \n    'prior_question_elapsed_time': 'float32', \n    'prior_question_had_explanation': 'bool'\n}\ntarget = 'answered_correctly'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = dt.fread('../input/riiid-test-answer-prediction/train.csv',\n                    columns=set(data_types_dict.keys())).to_pandas()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Train size: ',train_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#checking how much memory this dataframe is using\ntrain_df.memory_usage(deep=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#changing prior_question_had_explanation from object to boolean\ntrain_df['prior_question_had_explanation']=train_df['prior_question_had_explanation'].astype('boolean')\n\ntrain_df.memory_usage(deep=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nexample_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\nexample_sample_submission = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#number of unique users in our dataset\n\ntrain_df['user_id'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#unique user interactions\ntrain_df['content_id'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#unique user interactions which are questions\nprint(f\"We have {train_df['content_id'].nunique()} content ids of which {train_df[train_df['content_type_id']==False]['content_id'].nunique()} are questions \")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['task_container_id'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\ntrain_df['answered_correctly'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Timestamp\ntimestamp is important because it is user interaction and the first event from that user. so starting\ntime could be different for each user"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(train_df['timestamp'], bins=40);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby(['user_id'])['timestamp'].max().sort_values(ascending=False).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"Feature engineering"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=train_df.loc[train_df['answered_correctly']!=-1].reset_index(drop=True)\ntrain_df=train_df.drop(['timestamp','content_type_id'], axis=1)\ntrain_df['prior_question_had_explanation']=train_df['prior_question_had_explanation'].fillna(value=False).astype(bool)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_answers_df=train_df.groupby('user_id').agg({'answered_correctly': ['mean', 'count']}).copy()\nuser_answers_df.columns=['mean_user_accuracy','questions_answered']\n\ncontent_answers_df =train_df.groupby('content_id').agg({'answered_correctly':['mean','count']}).copy()\ncontent_answers_df.columns=['mean_accuracy','question_asked']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df.iloc[90000000:,:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=train_df.merge(user_answers_df, how='left', on='user_id')\ntrain_df=train_df.merge(content_answers_df, how='left', on='content_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.fillna(value=0.5, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"le = LabelEncoder()\ntrain_df[\"prior_question_had_explanation\"] = le.fit_transform(train_df[\"prior_question_had_explanation\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=train_df.sort_values(['user_id'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y=train_df['answered_correctly']\n\ncolumns = ['mean_user_accuracy', 'questions_answered', 'mean_accuracy', 'question_asked',\n           'prior_question_had_explanation']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=train_df[columns]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scores=[]\nfeature_importance=pd.DataFrame()\nmodels=[]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {'num_leaves': 32,\n          'max_bin': 300,\n          'objective': 'binary',\n          'max_depth': 13,\n          'learning_rate': 0.03,\n          \"boosting_type\": \"gbdt\",\n          \"metric\": 'auc',\n         }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns = ['mean_user_accuracy', 'questions_answered', 'mean_accuracy', 'question_asked',\n#            'prior_question_had_explanation', 'mean_diff1', 'mean_diff2'\n          ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"folds = StratifiedKFold(n_splits=5, shuffle=False)\nfor fold_n, (train_index, valid_index) in enumerate(folds.split(X, y)):\n    print(f'Fold {fold_n} started at {time.ctime()}')\n    X_train, X_valid = X[columns].iloc[train_index], X[columns].iloc[valid_index]\n    y_train, y_valid = y.iloc[train_index], y.iloc[valid_index]\n    model = lgb.LGBMClassifier(**params, n_estimators=700, n_jobs = 1)\n    model.fit(X_train, y_train, \n            eval_set=[(X_train, y_train), (X_valid, y_valid)],eval_metric='auc',verbose=1000, early_stopping_rounds=10)\n    score = max(model.evals_result_['valid_1']['auc'])\n    \n    models.append(model)\n    scores.append(score)\n\n    fold_importance = pd.DataFrame()\n    fold_importance[\"feature\"] = columns\n    fold_importance[\"importance\"] = model.feature_importances_\n    fold_importance[\"fold\"] = fold_n + 1\n    feature_importance = pd.concat([feature_importance, fold_importance], axis=0)\n    break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('CV mean score: {0:.4f}, std: {1:.4f}.'.format(np.mean(scores), np.std(scores)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_importance[\"importance\"] /= 1\ncols = feature_importance[[\"feature\", \"importance\"]].groupby(\"feature\").mean().sort_values(\n    by=\"importance\", ascending=False)[:50].index\n\nbest_features = feature_importance.loc[feature_importance.feature.isin(cols)]\n\nplt.figure(figsize=(16, 12));\nsns.barplot(x=\"importance\", y=\"feature\", data=best_features.sort_values(by=\"importance\", ascending=False));\nplt.title('LGB Features (avg over folds)');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del X,y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"iter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    y_preds = []\n    test_df = test_df.merge(user_answers_df, how = 'left', on = 'user_id')\n    test_df = test_df.merge(content_answers_df, how = 'left', on = 'content_id')\n    test_df['prior_question_had_explanation'] = test_df['prior_question_had_explanation'].fillna(value = False).astype(bool)\n    test_df = test_df.loc[test_df['content_type_id'] == 0].reset_index(drop=True)\n    test_df.fillna(value = 0.5, inplace = True)\n    test_df[\"prior_question_had_explanation_enc\"] = le.fit_transform(test_df[\"prior_question_had_explanation\"])\n    for model in models:\n        y_pred = model.predict_proba(test_df[columns], num_iteration=model.best_iteration_)[:, 1]\n        y_preds.append(y_pred)\n\n    y_preds = sum(y_preds) / len(y_preds)\n    test_df['answered_correctly'] = y_preds\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}