{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline: Preserve Train Proportions\n\nIt is always important to establish a simple baseline before building more sophisticated models. This way you have a reference to assess the true value added of your complex models. Often, it's hard to outperform these simple baseline models or the added complexity of other models is hard to justify.\n\nSo lets establish a simple baseline model that does random predictions that preserve the proportion of class predictions from the training data. This is of course not a very useful model but gives us something to compare to.","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"!pip install catboost\n!pip install --upgrade category_encoders\n!pip install optuna","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:41:03.901282Z","iopub.execute_input":"2023-02-27T13:41:03.901701Z","iopub.status.idle":"2023-02-27T13:44:41.599438Z","shell.execute_reply.started":"2023-02-27T13:41:03.90161Z","shell.execute_reply":"2023-02-27T13:44:41.598473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nimport pandas as pd\nimport seaborn as sns\nimport jo_wilder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-27T13:44:41.601096Z","iopub.execute_input":"2023-02-27T13:44:41.601394Z","iopub.status.idle":"2023-02-27T13:44:42.152627Z","shell.execute_reply.started":"2023-02-27T13:44:41.601363Z","shell.execute_reply":"2023-02-27T13:44:42.151627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gensim\nimport os\n# import polars as pl\nfrom tqdm.notebook import tqdm\npd.set_option(\"display.max_columns\", 999)\nfrom sklearn.pipeline import Pipeline\nimport optuna\nfrom sklearn.ensemble import RandomForestRegressor\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostClassifier, Pool\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom xgboost import XGBClassifier\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:44:42.156212Z","iopub.execute_input":"2023-02-27T13:44:42.156517Z","iopub.status.idle":"2023-02-27T13:44:43.256521Z","shell.execute_reply.started":"2023-02-27T13:44:42.156489Z","shell.execute_reply":"2023-02-27T13:44:43.255717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data\nHere we load the training label data so we can see what the most common output class.","metadata":{}},{"cell_type":"code","source":"# load training lables\ntrain_df = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:44:43.260221Z","iopub.execute_input":"2023-02-27T13:44:43.260925Z","iopub.status.idle":"2023-02-27T13:44:43.526016Z","shell.execute_reply.started":"2023-02-27T13:44:43.260897Z","shell.execute_reply":"2023-02-27T13:44:43.525257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Class proportion\nHere we have a look to see what the most common class (whether the user answered the quistion correctly). The assumption is that the training data is representative of the test data as well. If this holds true, there would be a similar proportion of this class in the test data.","metadata":{}},{"cell_type":"code","source":"GROUP_MAPPING = {\n    'q1': '0-4',\n    'q2': '0-4',\n    'q3': '0-4',\n    'q4': '5-12',\n    'q5': '5-12',\n    'q6': '5-12',\n    'q7': '5-12',\n    'q8': '5-12',\n    'q9': '5-12',\n    'q10': '5-12',\n    'q11': '5-12',\n    'q12': '5-12',\n    'q13': '5-12',\n    'q14': '13-22',\n    'q15': '13-22',\n    'q16': '13-22',\n    'q17': '13-22',\n    'q18': '13-22'\n}\n\nNUMERIC_DF_TYPES = ['int8', 'int16', 'int32', 'int64',\n                    'uint8', 'uint16', 'uint32', 'uint64',\n                    'float16', 'float32', 'float64']\n\naggs_list = ['min','max', 'mean', 'median', 'count', 'sum']\ndrop_cols = ['level_min','level_max','index_min','elapsed_time_min']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:44:43.52718Z","iopub.execute_input":"2023-02-27T13:44:43.52763Z","iopub.status.idle":"2023-02-27T13:44:43.533755Z","shell.execute_reply.started":"2023-02-27T13:44:43.527602Z","shell.execute_reply":"2023-02-27T13:44:43.533061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_grouping(df, aggs_list, drop_cols=[]):\n    '''\n    '''\n    try:\n        df = df.drop(columns=['session_level'])\n        print('TEST data preparation...')\n    except:\n        print('TRAIN data preparation...')\n    \n    df = df.drop(columns=['music','hq','fullscreen'])\n    num_feature = list(df.select_dtypes(include=NUMERIC_DF_TYPES).columns)\n    aggs_mapping = {k:aggs_list for k in (set(num_feature)-set(['session_id']))}\n\n    df_group = df.groupby(['session_id','level_group']).agg(aggs_mapping)\n    df_group.columns = ['_'.join(col) for col in df_group.columns.values]\n    df_group = df_group.reset_index()\n\n    df_group = df_group.drop(columns=drop_cols)\n\n    return df_group\n\n\ndef combine_train(df_group, df_labels):\n    '''\n    '''\n\n    df_labels = df_labels.rename(columns={'session_id':'session_id_row'})\n    df_labels['n_question'] = df_labels['session_id_row'].apply(lambda x: x.split('_')[1])\n    df_labels['session_id'] = df_labels['session_id_row'].apply(lambda x: x.split('_')[0])\n    df_labels['level_group'] = df_labels['n_question'].map(GROUP_MAPPING)\n    df_labels = df_labels.astype({'session_id':'int64'})\n\n    df_merged = df_group.merge(df_labels,on=['session_id','level_group'])\n\n    df_merged = df_merged.drop(columns=['session_id_row'])\n    return df_merged\n\ndef test_prep(df_group_test):\n    '''\n    '''\n    \n    group_mapping2 = {\n        '0-4': ['q1','q2','q3'],\n        '5-12': ['q4','q5','q6','q7','q8','q9','q10','q11','q12','q13'],\n        '13-22': ['q14','q15','q16','q17','q18']\n    }\n\n    n= []\n    for i in group_mapping2:\n        for v in group_mapping2[i]:\n            for b in list(df_group_test.session_id.unique()):\n                n.append([i,v,b])\n    df_pregroup2 = pd.DataFrame(n, columns = ['level_group','n_question','session_id'])\n    df_test_merged = df_group_test.merge(df_pregroup2,on=['session_id','level_group'])\n\n    return df_test_merged","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:44:43.534859Z","iopub.execute_input":"2023-02-27T13:44:43.535304Z","iopub.status.idle":"2023-02-27T13:44:43.548564Z","shell.execute_reply.started":"2023-02-27T13:44:43.535266Z","shell.execute_reply":"2023-02-27T13:44:43.547818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ndf_labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ndf_test = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:44:43.549696Z","iopub.execute_input":"2023-02-27T13:44:43.550152Z","iopub.status.idle":"2023-02-27T13:45:40.57637Z","shell.execute_reply.started":"2023-02-27T13:44:43.550123Z","shell.execute_reply":"2023-02-27T13:45:40.575407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_group = data_grouping(df, aggs_list, drop_cols)\ndf_merged_train = combine_train(df_group, df_labels)\ndf_group_test = data_grouping(df_test,aggs_list,drop_cols)\ndf_merged_test = test_prep(df_group_test)\ndf_merged_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:45:40.577482Z","iopub.execute_input":"2023-02-27T13:45:40.577852Z","iopub.status.idle":"2023-02-27T13:45:53.274069Z","shell.execute_reply.started":"2023-02-27T13:45:40.577825Z","shell.execute_reply":"2023-02-27T13:45:53.273344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged_train[:18000]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:45:53.274999Z","iopub.execute_input":"2023-02-27T13:45:53.275526Z","iopub.status.idle":"2023-02-27T13:45:53.342282Z","shell.execute_reply.started":"2023-02-27T13:45:53.275498Z","shell.execute_reply":"2023-02-27T13:45:53.341394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_merged_train[df_merged_train['level_group']=='0-4']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:45:53.345102Z","iopub.execute_input":"2023-02-27T13:45:53.34536Z","iopub.status.idle":"2023-02-27T13:45:53.348292Z","shell.execute_reply.started":"2023-02-27T13:45:53.345336Z","shell.execute_reply":"2023-02-27T13:45:53.347682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_0_4 = df_merged_train[df_merged_train['level_group']=='0-4']\ntrain_df_5_12 = df_merged_train[df_merged_train['level_group']=='5-12']\ntrain_df_13_22 = df_merged_train[df_merged_train['level_group']=='13-22']","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:45:53.349379Z","iopub.execute_input":"2023-02-27T13:45:53.349893Z","iopub.status.idle":"2023-02-27T13:45:53.456147Z","shell.execute_reply.started":"2023-02-27T13:45:53.349867Z","shell.execute_reply":"2023-02-27T13:45:53.455403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GroupShuffleSplit\n\n# TARGET = 'correct'\n# CAT_FEATURES = ['n_question','level_group']\n\n# params = {\n#     'n_estimators': 1000,\n#     'depth': 9,\n#     'verbose': 200,\n#     'eval_metric': 'AUC'\n#     }\n\n# gs = GroupShuffleSplit(n_splits=2, test_size=.3, random_state=0)\n# # train_idx, test_idx = next(gs.split(df_merged_train[:9000].loc[:,df_merged_train.columns != TARGET], df_merged_train[:9000][TARGET], groups=df_merged_train[:9000].session_id))\n# train_idx, test_idx = next(gs.split(df_merged_train.loc[:,df_merged_train.columns != TARGET], df_merged_train[TARGET], groups=df_merged_train.session_id))\n\n# train = df_merged_train.iloc[train_idx]\n# test = df_merged_train.iloc[test_idx]\n\n# model = CatBoostClassifier(**params)\n\n# train_pool = Pool(train.loc[:,train.columns != TARGET],\n#                   train[TARGET], cat_features=CAT_FEATURES)\n# test_pool = Pool(test.loc[:,test.columns != TARGET],\n#                   test[TARGET], cat_features=CAT_FEATURES)\n\n# model_0_4.fit(train_pool, eval_set=test_pool,\n#           early_stopping_rounds=100,\n#             use_best_model=True)\n\n# # preds = model.predict(test_pool)\n\n# # print(classification_report(test.loc[:,test.columns == TARGET], preds))","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:45:53.457319Z","iopub.execute_input":"2023-02-27T13:45:53.457787Z","iopub.status.idle":"2023-02-27T13:45:53.462105Z","shell.execute_reply.started":"2023-02-27T13:45:53.457759Z","shell.execute_reply":"2023-02-27T13:45:53.461351Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupShuffleSplit\n\nTARGET = 'correct'\nCAT_FEATURES = ['n_question','level_group']\n\nparams = {\n    'n_estimators': 1000,\n    'depth': 9,\n    'verbose': 200,\n    'eval_metric': 'AUC'\n    }\n\ngs = GroupShuffleSplit(n_splits=2, test_size=.3, random_state=0)\n# train_idx, test_idx = next(gs.split(df_merged_train[:9000].loc[:,df_merged_train.columns != TARGET], df_merged_train[:9000][TARGET], groups=df_merged_train[:9000].session_id))\ntrain_idx, test_idx = next(gs.split(train_df_0_4.loc[:,train_df_0_4.columns != TARGET], train_df_0_4[TARGET], groups=train_df_0_4.session_id))\n\ntrain = train_df_0_4.iloc[train_idx]\ntest = train_df_0_4.iloc[test_idx]\n\nmodel_0_4 = CatBoostClassifier(**params)\n\ntrain_pool = Pool(train.loc[:,train.columns != TARGET],\n                  train[TARGET], cat_features=CAT_FEATURES)\ntest_pool = Pool(test.loc[:,test.columns != TARGET],\n                  test[TARGET], cat_features=CAT_FEATURES)\n\nmodel_0_4.fit(train_pool, eval_set=test_pool,\n          early_stopping_rounds=100,\n            use_best_model=True)\n\n# preds = model.predict(test_pool)\n\n# print(classification_report(test.loc[:,test.columns == TARGET], preds))","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:45:53.463247Z","iopub.execute_input":"2023-02-27T13:45:53.463527Z","iopub.status.idle":"2023-02-27T13:46:22.784086Z","shell.execute_reply.started":"2023-02-27T13:45:53.4635Z","shell.execute_reply":"2023-02-27T13:46:22.78341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gs = GroupShuffleSplit(n_splits=2, test_size=.3, random_state=0)\n# train_idx, test_idx = next(gs.split(df_merged_train[:9000].loc[:,df_merged_train.columns != TARGET], df_merged_train[:9000][TARGET], groups=df_merged_train[:9000].session_id))\ntrain_idx, test_idx = next(gs.split(train_df_5_12.loc[:,train_df_5_12.columns != TARGET], train_df_5_12[TARGET], groups=train_df_5_12.session_id))\n\ntrain = train_df_5_12.iloc[train_idx]\ntest = train_df_5_12.iloc[test_idx]\n\nmodel_5_12 = CatBoostClassifier(**params)\n\ntrain_pool = Pool(train.loc[:,train.columns != TARGET],\n                  train[TARGET], cat_features=CAT_FEATURES)\ntest_pool = Pool(test.loc[:,test.columns != TARGET],\n                  test[TARGET], cat_features=CAT_FEATURES)\n\nmodel_5_12.fit(train_pool, eval_set=test_pool,\n          early_stopping_rounds=100,\n            use_best_model=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:46:22.785167Z","iopub.execute_input":"2023-02-27T13:46:22.785586Z","iopub.status.idle":"2023-02-27T13:46:57.408145Z","shell.execute_reply.started":"2023-02-27T13:46:22.78556Z","shell.execute_reply":"2023-02-27T13:46:57.407315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gs = GroupShuffleSplit(n_splits=2, test_size=.3, random_state=0)\n# train_idx, test_idx = next(gs.split(df_merged_train[:9000].loc[:,df_merged_train.columns != TARGET], df_merged_train[:9000][TARGET], groups=df_merged_train[:9000].session_id))\ntrain_idx, test_idx = next(gs.split(train_df_13_22.loc[:,train_df_13_22.columns != TARGET], train_df_13_22[TARGET], groups=train_df_13_22.session_id))\n\ntrain = train_df_13_22.iloc[train_idx]\ntest = train_df_13_22.iloc[test_idx]\n\nmodel_13_22 = CatBoostClassifier(**params)\n\ntrain_pool = Pool(train.loc[:,train.columns != TARGET],\n                  train[TARGET], cat_features=CAT_FEATURES)\ntest_pool = Pool(test.loc[:,test.columns != TARGET],\n                  test[TARGET], cat_features=CAT_FEATURES)\n\nmodel_13_22.fit(train_pool, eval_set=test_pool,\n          early_stopping_rounds=100,\n            use_best_model=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:46:57.409429Z","iopub.execute_input":"2023-02-27T13:46:57.410007Z","iopub.status.idle":"2023-02-27T13:47:27.913506Z","shell.execute_reply.started":"2023-02-27T13:46:57.409973Z","shell.execute_reply":"2023-02-27T13:47:27.91253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds[15:18]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.914512Z","iopub.execute_input":"2023-02-27T13:47:27.91481Z","iopub.status.idle":"2023-02-27T13:47:27.921559Z","shell.execute_reply.started":"2023-02-27T13:47:27.914783Z","shell.execute_reply":"2023-02-27T13:47:27.92073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make \"predictions\"\nAs you can see, around 70% of the quiestions in the training data was answered correctly. We could simply always predict `1` with this probability and `0` otherwise.","metadata":{}},{"cell_type":"code","source":"# d = {'session_id': ['20090109393214576_q1', '20090109393214576_q1','20090109393214576_q1'], 'correct': [0, 0,0]}\n# sample_submission= pd.DataFrame(data=d, index=[0, 1, 2])","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.924721Z","iopub.execute_input":"2023-02-27T13:47:27.9251Z","iopub.status.idle":"2023-02-27T13:47:27.930494Z","shell.execute_reply.started":"2023-02-27T13:47:27.925072Z","shell.execute_reply":"2023-02-27T13:47:27.92959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_submission[\"correct\"] = preds[15:18]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.931906Z","iopub.execute_input":"2023-02-27T13:47:27.932219Z","iopub.status.idle":"2023-02-27T13:47:27.941539Z","shell.execute_reply.started":"2023-02-27T13:47:27.932184Z","shell.execute_reply":"2023-02-27T13:47:27.940729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_merged_test[:1].level_group[0]","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.942733Z","iopub.execute_input":"2023-02-27T13:47:27.943695Z","iopub.status.idle":"2023-02-27T13:47:27.951067Z","shell.execute_reply.started":"2023-02-27T13:47:27.943644Z","shell.execute_reply":"2023-02-27T13:47:27.950201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if df_merged_test[:1].level_group[0] == '13-22':\n#  print(1)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.952397Z","iopub.execute_input":"2023-02-27T13:47:27.952697Z","iopub.status.idle":"2023-02-27T13:47:27.961026Z","shell.execute_reply.started":"2023-02-27T13:47:27.95265Z","shell.execute_reply":"2023-02-27T13:47:27.960188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.962636Z","iopub.execute_input":"2023-02-27T13:47:27.962978Z","iopub.status.idle":"2023-02-27T13:47:27.975095Z","shell.execute_reply.started":"2023-02-27T13:47:27.962947Z","shell.execute_reply":"2023-02-27T13:47:27.97424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\nprint(\"Predictions:\")\nfor (sample_submission, test) in iter_test:\n    df_group_test = data_grouping(test,aggs_list,drop_cols)\n    df_merged_test = test_prep(df_group_test)\n    if df_merged_test[:1].level_group[0] == '0-4':\n        sample_submission[\"correct\"] = model_0_4.predict(df_merged_test)\n    elif df_merged_test[:1].level_group[0] == '5-12':\n            sample_submission[\"correct\"] = model_5_12.predict(df_merged_test)\n    elif df_merged_test[:1].level_group[0] == '13-22':\n            sample_submission[\"correct\"] = model_13_22.predict(df_merged_test)\n# #     sample_submission[\"correct\"] = model.predict(df_merged_test)\n#     print(sample_submission)\n#     print(df_merged_test)\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T13:47:27.97653Z","iopub.execute_input":"2023-02-27T13:47:27.977093Z","iopub.status.idle":"2023-02-27T13:47:28.376044Z","shell.execute_reply.started":"2023-02-27T13:47:27.977057Z","shell.execute_reply":"2023-02-27T13:47:28.375311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}