{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.metrics import roc_auc_score\n\n\nfrom catboost import CatBoostClassifier, Pool\n\n\nfrom matplotlib import ticker\nimport time\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.metrics import f1_score\nimport gc\nimport random","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:19.572256Z","iopub.execute_input":"2023-05-01T07:54:19.573306Z","iopub.status.idle":"2023-05-01T07:54:20.249470Z","shell.execute_reply.started":"2023-05-01T07:54:19.573227Z","shell.execute_reply":"2023-05-01T07:54:20.248419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # READ USER ID ONLY\ntype_dic = {'session_id': 'category',\n          'elapsed_time': np.int32,\n          'event_name': 'category',\n          'name': 'category',\n          'level': np.uint8,\n          'page': 'category',\n          'room_coor_x': np.float32,\n          'room_coor_y': np.float32,\n          'screen_coor_x': np.float32,\n          'screen_coor_y': np.float32,\n          'hover_duration': np.float32,\n          'text': 'category',\n          'fqid': 'category',\n          'room_fqid': 'category',\n          'text_fqid': 'category',\n          'fullscreen': np.int8,\n          'hq': np.int8,\n          'music': np.int8,\n          'level_group': 'category'}\n# 本番はちゃんとnrowsを消す！\ntrain_df = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", dtype=type_dic)\ntrain_df.head(10)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-01T07:54:20.254501Z","iopub.execute_input":"2023-05-01T07:54:20.255123Z","iopub.status.idle":"2023-05-01T07:54:23.615555Z","shell.execute_reply.started":"2023-05-01T07:54:20.255086Z","shell.execute_reply":"2023-05-01T07:54:23.614441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\n# train_labelsの確認\ntest_df = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:23.616790Z","iopub.execute_input":"2023-05-01T07:54:23.617805Z","iopub.status.idle":"2023-05-01T07:54:23.972167Z","shell.execute_reply.started":"2023-05-01T07:54:23.617744Z","shell.execute_reply":"2023-05-01T07:54:23.970426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# セッションidとquestion番号がくっついてるから分けてあげる．\ntrain_label['session'] = train_label.session_id.apply(lambda x: str(x.split('_')[0]) )\ntrain_label['q'] = train_label.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( 'shape of label dataset is:',train_label.shape )\ntrain_label.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:23.975182Z","iopub.execute_input":"2023-05-01T07:54:23.975873Z","iopub.status.idle":"2023-05-01T07:54:24.592928Z","shell.execute_reply.started":"2023-05-01T07:54:23.975829Z","shell.execute_reply":"2023-05-01T07:54:24.591863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\n# NUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n#         'screen_coor_x', 'screen_coor_y', 'hover_duration']\nNUMS = ['elapsed_time','level']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:24.600115Z","iopub.execute_input":"2023-05-01T07:54:24.600780Z","iopub.status.idle":"2023-05-01T07:54:24.612098Z","shell.execute_reply.started":"2023-05-01T07:54:24.600744Z","shell.execute_reply":"2023-05-01T07:54:24.611044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:24.613380Z","iopub.execute_input":"2023-05-01T07:54:24.614249Z","iopub.status.idle":"2023-05-01T07:54:24.625401Z","shell.execute_reply.started":"2023-05-01T07:54:24.614214Z","shell.execute_reply":"2023-05-01T07:54:24.624250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr = feature_engineer(train_df)\ndf_tr.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:24.627143Z","iopub.execute_input":"2023-05-01T07:54:24.627524Z","iopub.status.idle":"2023-05-01T07:54:26.100079Z","shell.execute_reply.started":"2023-05-01T07:54:24.627488Z","shell.execute_reply":"2023-05-01T07:54:26.098853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df_tr.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\n\n# 1セッション1ユーザーだからセッション数=ユーザー数\nALL_USERS = df_tr.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:26.101785Z","iopub.execute_input":"2023-05-01T07:54:26.102161Z","iopub.status.idle":"2023-05-01T07:54:26.109438Z","shell.execute_reply.started":"2023-05-01T07:54:26.102127Z","shell.execute_reply":"2023-05-01T07:54:26.108027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\ncat_oof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df_tr, groups=df_tr.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    \n    lgb_params = {\n    'objective' : 'binary',\n    'metric' : 'auc',\n    'learning_rate': 0.002,\n    'max_depth': 6,\n    'num_iterations': 1000}\n    \n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df_tr.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_label.loc[train_label.q==t].set_index('session').loc[train_users]\n\n        train_pool = Pool(train_x[FEATURES].astype('float32'), train_y[\"correct\"])\n        \n        # VALID DATA\n        valid_x = df_tr.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_label.loc[train_label.q==t].set_index('session').loc[valid_users]\n\n        valid_pool = Pool(valid_x[FEATURES].astype('float32'), valid_y[\"correct\"])\n        \n        # TRAIN MODEL\n        model = CatBoostClassifier(\n            iterations = 1000,\n            early_stopping_rounds = 50,\n            depth = 4,\n            learning_rate = 0.05,\n            loss_function = \"Logloss\",\n            random_seed = 0,\n            metric_period = 1,\n            subsample = 0.8,\n            colsample_bylevel = 0.4,\n            verbose = 0,\n            task_type=\"CPU\"\n        )\n        model = model.fit(train_pool, eval_set = valid_pool)\n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'fold{i}_question{t}'] = model\n        print(model.predict_proba(valid_pool)[:, 1][0])\n        cat_oof.loc[valid_users, t-1] = model.predict_proba(valid_pool)[:,1]\n        \n    print()","metadata":{"papermill":{"duration":69.877213,"end_time":"2023-02-07T01:02:43.57299","exception":false,"start_time":"2023-02-07T01:01:33.695777","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-01T07:54:26.111352Z","iopub.execute_input":"2023-05-01T07:54:26.111803Z","iopub.status.idle":"2023-05-01T07:54:43.098518Z","shell.execute_reply.started":"2023-05-01T07:54:26.111765Z","shell.execute_reply":"2023-05-01T07:54:43.097378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = cat_oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_label.loc[train_label.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:43.100049Z","iopub.execute_input":"2023-05-01T07:54:43.100403Z","iopub.status.idle":"2023-05-01T07:54:43.417910Z","shell.execute_reply.started":"2023-05-01T07:54:43.100371Z","shell.execute_reply":"2023-05-01T07:54:43.416542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (cat_oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:43.419794Z","iopub.execute_input":"2023-05-01T07:54:43.420248Z","iopub.status.idle":"2023-05-01T07:54:43.719313Z","shell.execute_reply.started":"2023-05-01T07:54:43.420213Z","shell.execute_reply":"2023-05-01T07:54:43.717690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:43.722946Z","iopub.execute_input":"2023-05-01T07:54:43.723326Z","iopub.status.idle":"2023-05-01T07:54:43.731546Z","shell.execute_reply.started":"2023-05-01T07:54:43.723294Z","shell.execute_reply":"2023-05-01T07:54:43.730031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 以下テスト\n* models, best_thresholdが必要","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:43.733410Z","iopub.execute_input":"2023-05-01T07:54:43.733918Z","iopub.status.idle":"2023-05-01T07:54:43.747177Z","shell.execute_reply.started":"2023-05-01T07:54:43.733875Z","shell.execute_reply":"2023-05-01T07:54:43.745723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (test, sample_submission) in iter_test:\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    \n    if counter == 0:\n        print(sample_submission.head())\n        print(test.head())\n        print(test.shape)\n    \n    sample_submission['question'] = [int(label.split('_')[1][1:]) for label in sample_submission['session_id']]\n    \n    \n    # columnsを指定してあげないとevent_nameに入っていないeventのカラムができない．\n    test = feature_engineer(test)\n    \n    # level_group以外の特徴量, FEATURES\n    test_pool = Pool(test[FEATURES].astype('float32'))\n    \n    # questionごとの予測値を格納\n    for q in range(a,b):\n        mask = sample_submission.question == q\n        try:\n            pred = np.zeros(5)\n            for f in range(5):\n                model = models[f\"fold{f}_question{q}\"]\n                pred[f] = model.predict_proba(test_pool)[:, 1][0]\n            pred = np.mean(pred, keepdims=True)\n            sample_submission.loc[mask,'correct'] = (pred > best_threshold).astype(int)\n        except:\n            sample_submission.loc[mask,'correct'] = 0\n    \n    ## submission\n    sample_submission = sample_submission.fillna(0)\n    env.predict(sample_submission[['session_id', 'correct']])\n    counter += 1","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:43.748893Z","iopub.execute_input":"2023-05-01T07:54:43.749558Z","iopub.status.idle":"2023-05-01T07:54:44.294272Z","shell.execute_reply.started":"2023-05-01T07:54:43.749514Z","shell.execute_reply":"2023-05-01T07:54:44.293263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## the end result is a submission file containing all test session predictions\n! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-05-01T07:54:44.295629Z","iopub.execute_input":"2023-05-01T07:54:44.296221Z","iopub.status.idle":"2023-05-01T07:54:45.405095Z","shell.execute_reply.started":"2023-05-01T07:54:44.296184Z","shell.execute_reply":"2023-05-01T07:54:45.403608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}