{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import gc\nimport matplotlib.pyplot as plt\n\n%matplotlib inline\n\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import StandardScaler\n\nimport lightgbm as lgb\nss = StandardScaler()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Preprocessing And Exploring Data"},{"metadata":{},"cell_type":"markdown","source":"There are awsome EDA notebooks out there, I am sharing the links here:\n\n- https://www.kaggle.com/isaienkov/riiid-answer-correctness-prediction-eda-modeling by **Isaienkov**\n- https://www.kaggle.com/ilialar/simple-eda-and-baseline by **Ilia**\n- https://www.kaggle.com/erikbruin/riiid-eda-of-full-dataset by **Erik**\n\nBased on these superb notebooks, I have also tried to explore the data more and my findings and exloration can be found [here](https://www.kaggle.com/mrutyunjaybiswal/riiid-aied-challenge-2020-data-exploration-and-fe).\n\nSo, I won't do more of exploration here, this notebook covers only feature engineering and modelling part based on the takeaways from above notebooks."},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv',\n                   usecols=[1, 2, 3, 4, 5, 7, 8, 9],\n                   dtype={'timestamp': 'int64',\n                          'user_id': 'int32',\n                          'content_id': 'int16',\n                          'content_type_id': 'int8',\n                          'task_container_id': 'int16',\n                          'answered_correctly':'int8',\n                          'prior_question_elapsed_time': 'float32',\n                          'prior_question_had_explanation': 'boolean'},\n                    nrows=1e7\n                   )\nquestions_df = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures_df = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')\n\ntrain","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.memory_usage()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Lectures.csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures_df.loc[:, \"type_of\"] = lectures_df[\"type_of\"].replace(\"solving question\", \"solving_question\")\nlectures_df = pd.get_dummies(lectures_df, columns=[\"type_of\", \"part\"])\nlect_part_cols = [col for col in lectures_df.columns if col.startswith(\"part\")]\nlect_type_cols = [col for col in lectures_df.columns if col.startswith(\"type_of_\")]\nlectures_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_lec = train[train['content_type_id']==1].merge(lectures_df, left_on=\"content_id\", right_on=\"lecture_id\", how=\"left\")\ntrain_lec.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_lecStats = train_lec.groupby(\"user_id\")[lect_part_cols + lect_type_cols].sum()\n\nfor col in user_lecStats.columns:\n    bool_col = col + \"_bool\"\n    user_lecStats[bool_col] = (user_lecStats[col]).astype(int)\n    \ndel train_lec\ngc.collect()\nuser_lecStats.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Train.csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train['content_type_id']==0].sort_values(\"timestamp\").reset_index(drop=True)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"g1 = train.loc[:, [\"task_container_id\", \"user_id\"]].groupby(\"task_container_id\").agg(\"count\")\ng1.columns = ['avg_qstns']\ng2 = train.loc[:, [\"task_container_id\", \"user_id\"]].groupby(\"task_container_id\").agg(\"nunique\")\ng2.columns = ['avg_qstns']\ng3 = g1/g2\ng3['avg_qstns_seen'] = g3['avg_qstns'].cumsum()\ng3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del g1, g2\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tgm1 = train.loc[:, [\"user_id\", \"answered_correctly\"]].groupby(\"user_id\").agg(\"mean\")\ntgm2 = train.loc[:, [\"user_id\", \"prior_question_had_explanation\"]].groupby(\"user_id\").agg(\"mean\").astype(bool)\ntgm1.columns = ['avg_ac']\ntgm2.columns = ['avg_pqhe']\ntgm2.loc[:, \"avg_pqhe\"] = tgm2['avg_pqhe'].replace({True:1, False:0})\ndisplay(tgm1.head())\ndisplay(tgm2.head())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- `g1`: Number of users per task\n- `g2`: Unique users per task\n- `g3`: Avg users per task\n- `tgm1`: Avg questoins answered correctly per user\n- `tgm2`: Avg questions that had explanation per user"},{"metadata":{},"cell_type":"markdown","source":"## Questions.csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"questions_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.merge(train, questions_df, left_on=\"content_id\", right_on=\"question_id\", how=\"left\")\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rqf1 = train.loc[:, [\"question_id\", \"answered_correctly\"]].groupby(\"question_id\").agg(\"mean\")\nrqf2 = train.loc[:, [\"question_id\", \"answered_correctly\"]].groupby(\"question_id\").agg(\"count\")\nrqf1.columns = [\"question_percent\"]\nrqf2.columns = [\"count\"]\ndisplay(rqf1.head())\ndisplay(rqf2.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question_df1 = pd.merge(questions_df, rqf1, left_on=\"question_id\", right_on=\"question_id\", how=\"left\")\nquestion_df1 = pd.merge(question_df1, rqf2, left_on=\"question_id\", right_on=\"question_id\", how=\"left\")\nquestion_df1['question_percent'] = round(question_df1[\"question_percent\"], 5)\ndisplay(question_df1.head())\ndisplay(question_df1.tail())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question_df1['tags_count'] = question_df1[\"tags\"].apply(lambda x: len(str(x).split()))\nquestion_df1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Creating Validation\n\n- Creating the validaion using Most recent answers by User\n- Inpsiration taken from @takamotoki"},{"metadata":{"trusted":true},"cell_type":"code","source":"val = train.groupby(\"user_id\").tail(25)\ntrain = train[~train.index.isin(val.index)]\nlen(train) + len(val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prerpocessing_pipeline(df, is_test=False):\n    df = pd.merge(df, user_lecStats, on=[\"user_id\"], how=\"left\")\n    df = pd.merge(df, tgm1, on=[\"user_id\"], how=\"left\")\n    df = pd.merge(df, tgm2, on=[\"user_id\"], how=\"left\")\n    df = pd.merge(df, g3, on=[\"task_container_id\"], how=\"left\")\n    if not is_test:\n        df = pd.merge(df, question_df1, on=[\"question_id\"], how=\"left\")\n    else:\n        df = pd.merge(df, question_df1, left_on=\"content_id\", right_on=\"question_id\", how=\"left\")\n    \n    df.drop(columns=['tags_x', 'user_id', 'content_type_id', 'question_id', 'bundle_id_x', 'correct_answer_y', 'part_y', 'tags_y', 'bundle_id_y', 'content_id'], inplace=True)\n    df['prior_question_elapsed_time'].fillna(0, inplace=True)\n    df['prior_question_had_explanation'].fillna(0, inplace=True)\n    df['correct_answer_x'].fillna(2, inplace=True)\n    df['part_x'].fillna(2, inplace=True)\n    df['part_1'].fillna(0, inplace=True)\n    df['part_2'].fillna(0, inplace=True)\n    df['part_3'].fillna(0, inplace=True)\n    df['part_4'].fillna(0, inplace=True)\n    df['part_5'].fillna(0, inplace=True)\n    df['part_6'].fillna(0, inplace=True)\n    df['part_7'].fillna(0, inplace=True)\n    df['type_of_concept'].fillna(0, inplace=True)\n    df['type_of_intention'].fillna(0, inplace=True)\n    df['type_of_solving_question'].fillna(0, inplace=True)\n    df['type_of_starter'].fillna(0, inplace=True)\n    df['part_1_bool'].fillna(0, inplace=True)\n    df['part_2_bool'].fillna(0, inplace=True)\n    df['part_3_bool'].fillna(0, inplace=True)\n    df['part_4_bool'].fillna(0, inplace=True)\n    df['part_5_bool'].fillna(0, inplace=True)\n    df['part_6_bool'].fillna(0, inplace=True)\n    df['part_7_bool'].fillna(0, inplace=True)\n    df['type_of_concept_bool'].fillna(0, inplace=True)\n    df['type_of_intention_bool'].fillna(0, inplace=True)\n    df['type_of_solving_question_bool'].fillna(0, inplace=True)\n    df['type_of_starter_bool'].fillna(0, inplace=True)\n    df['timestamp'] = ss.fit_transform(df['timestamp'].values.reshape(-1, 1))\n    df['prior_question_elapsed_time'] = ss.fit_transform(df['prior_question_elapsed_time'].values.reshape(-1, 1))\n    df['count'] = ss.fit_transform(df['count'].values.reshape(-1, 1))\n    df['prior_question_had_explanation'] = df['prior_question_had_explanation'].replace({True:1, False:0})\n    \n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = prerpocessing_pipeline(train)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.rename(columns={\"part_x\":\"part\", \"correct_answer_x\":\"correct_answer\"})\nfeatures = train.columns.tolist()\nfeatures","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val = prerpocessing_pipeline(val)\nval.columns = features\nval.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# space free karo\ndel lectures_df, questions_df, rqf1, rqf2\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Modelling"},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train.drop(['answered_correctly'], axis=1)\ny = train['answered_correctly']\n\nX_val = val.drop(['answered_correctly'], axis=1)\ny_val = val['answered_correctly']\n\nX.shape, y.shape, X_val.shape, y_val.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del train, val\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cat_ftrs = ['task_container_id', 'correct_answer', 'part', 'tags_count']\nlgb_train = lgb.Dataset(X, y, categorical_feature=cat_ftrs)\ndel X, y\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lgb_eval = lgb.Dataset(X_val, y_val, categorical_feature=cat_ftrs, reference=lgb_train)\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features_finale = X_val.columns.tolist()\nlen(features_finale)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {\n    'objective': 'binary',\n    'boosting' : 'gbdt',\n    'max_bin': 800,\n    'learning_rate': 0.0175,\n    'num_leaves': 80\n}\n\nmodel = lgb.train(\n    params, lgb_train,\n    valid_sets=[lgb_train, lgb_eval],\n    verbose_eval=50,\n    num_boost_round=10000,\n    early_stopping_rounds=12\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = model.predict(X_val)\ny_true = np.array(y_val)\nroc_auc_score(y_true, y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lgb.plot_importance(model)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lgb.plot_importance(model, importance_type=\"gain\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"import riiideducation\n\nenv = riiideducation.make_env()\niter_test = env.iter_test()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for test_df, sample_prediction_df in iter_test:\n    \n    test_df = pd.merge(test_df, user_lecStats, on=[\"user_id\"], how=\"left\")\n    test_df = pd.merge(test_df, tgm1, on=[\"user_id\"], how=\"left\")\n    test_df = pd.merge(test_df, tgm2, on=[\"user_id\"], how=\"left\")\n    test_df = pd.merge(test_df, g3, on=[\"task_container_id\"], how=\"left\")\n    test_df = pd.merge(test_df, question_df1, left_on=\"content_id\", right_on=\"question_id\", how=\"left\")\n    test_df['prior_question_elapsed_time'].fillna(0, inplace=True)\n    test_df['prior_question_had_explanation'].fillna(0, inplace=True)\n    test_df['correct_answer'].fillna(2, inplace=True)\n    test_df['part'].fillna(2, inplace=True)\n    test_df['part_1'].fillna(0, inplace=True)\n    test_df['part_2'].fillna(0, inplace=True)\n    test_df['part_3'].fillna(0, inplace=True)\n    test_df['part_4'].fillna(0, inplace=True)\n    test_df['part_5'].fillna(0, inplace=True)\n    test_df['part_6'].fillna(0, inplace=True)\n    test_df['part_7'].fillna(0, inplace=True)\n    test_df['type_of_concept'].fillna(0, inplace=True)\n    test_df['type_of_intention'].fillna(0, inplace=True)\n    test_df['type_of_solving_question'].fillna(0, inplace=True)\n    test_df['type_of_starter'].fillna(0, inplace=True)\n    test_df['part_1_bool'].fillna(0, inplace=True)\n    test_df['part_2_bool'].fillna(0, inplace=True)\n    test_df['part_3_bool'].fillna(0, inplace=True)\n    test_df['part_4_bool'].fillna(0, inplace=True)\n    test_df['part_5_bool'].fillna(0, inplace=True)\n    test_df['part_6_bool'].fillna(0, inplace=True)\n    test_df['part_7_bool'].fillna(0, inplace=True)\n    test_df['type_of_concept_bool'].fillna(0, inplace=True)\n    test_df['type_of_intention_bool'].fillna(0, inplace=True)\n    test_df['type_of_solving_question_bool'].fillna(0, inplace=True)\n    test_df['type_of_starter_bool'].fillna(0, inplace=True)\n    test_df['timestamp'] = ss.fit_transform(test_df['timestamp'].values.reshape(-1, 1))\n    test_df['prior_question_elapsed_time'] = ss.fit_transform(test_df['prior_question_elapsed_time'].values.reshape(-1, 1))\n    test_df['count'] = ss.fit_transform(test_df['count'].values.reshape(-1, 1))\n    test_df['prior_question_had_explanation'] = test_df['prior_question_had_explanation'].replace({True:1, False:0})\n    \n    test_df['answered_correctly'] = model.predict(test_df[features_finale])\n    env.predict(test_df.loc[test_df['content_type_id'] == 0,\n               ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}