{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Catboost Baseline - LB 0.6**\n\nhttps://www.kaggle.com/code/cdeotte/xgboost-baseline-0-678\n\n上記をベースにCatboostを適用する","metadata":{"papermill":{"duration":0.005932,"end_time":"2023-02-07T00:59:58.147501","exception":false,"start_time":"2023-02-07T00:59:58.141569","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n#import cudf as cu\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\n\n#追加(CatBoost用)\nfrom catboost import CatBoost, CatBoostRegressor, CatBoostClassifier\nfrom catboost import Pool\n\npd.set_option(\"display.max_columns\", 1000)\npd.set_option(\"display.max_rows\", 1000)\n\n#追加(グラフ作成用)\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"papermill":{"duration":1.027875,"end_time":"2023-02-07T00:59:59.180261","exception":false,"start_time":"2023-02-07T00:59:58.152386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:05:22.095861Z","iopub.execute_input":"2023-02-12T12:05:22.096405Z","iopub.status.idle":"2023-02-12T12:05:23.631671Z","shell.execute_reply.started":"2023-02-12T12:05:22.096287Z","shell.execute_reply":"2023-02-12T12:05:23.630601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reducing the amount of RAM consumption","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# こちらを参考　https://www.kaggle.com/code/mohammad2012191/reduce-memory-usage-2gb-780mb\n# 'category'はエラーになるので'object'に変更\ndef reduce_mem_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage of dataframe is {start_mem:.2f} MB')\n    for col in df.columns:\n        if df[col].dtype == 'object':\n            df[col] = df[col].astype('object')\n        elif df[col].dtype == 'int':\n            int_types = [np.int8, np.int16, np.int32, np.int64]\n            for int_type in int_types:\n                if df[col].min() >= np.iinfo(int_type).min and df[col].max() <= np.iinfo(int_type).max:\n                    df[col] = df[col].astype(int_type)\n                    break\n        elif df[col].dtype == 'float':\n            float_types = [np.float16, np.float32, np.float64]\n            for float_type in float_types:\n                if df[col].min() >= np.finfo(float_type).min and df[col].max() <= np.finfo(float_type).max:\n                    df[col] = df[col].astype(float_type)\n                    break\n    mem_usage = df.memory_usage().sum() / 1024**2 \n    print(f\"Memory usage became: {mem_usage:.2f} MB\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:05:23.633466Z","iopub.execute_input":"2023-02-12T12:05:23.633817Z","iopub.status.idle":"2023-02-12T12:05:23.644589Z","shell.execute_reply.started":"2023-02-12T12:05:23.633787Z","shell.execute_reply":"2023-02-12T12:05:23.643532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# こちらを参考　https://www.kaggle.com/code/curiosity30/reduce-memory-2gb-500mb\ntrain = reduce_mem_usage(pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv'))\ntrain_cols = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page', \\\n              'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \\\n              'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']\nprint( train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:05:23.645949Z","iopub.execute_input":"2023-02-12T12:05:23.646259Z","iopub.status.idle":"2023-02-12T12:06:25.736558Z","shell.execute_reply.started":"2023-02-12T12:05:23.646232Z","shell.execute_reply":"2023-02-12T12:06:25.735697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 全てNaNなのでDropする\ntrain = train.drop(['fullscreen','hq','music'],axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = reduce_mem_usage(pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv'))\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"papermill":{"duration":0.598155,"end_time":"2023-02-07T01:00:59.082015","exception":false,"start_time":"2023-02-07T01:00:58.48386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:06:25.738775Z","iopub.execute_input":"2023-02-12T12:06:25.739261Z","iopub.status.idle":"2023-02-12T12:06:26.336556Z","shell.execute_reply.started":"2023-02-12T12:06:25.739232Z","shell.execute_reply":"2023-02-12T12:06:26.335439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\n基本的な集計機能を作成します。CVやLBを上げるために、より多くの機能を作ってみてください EVENTS機能のアイデアは[こちら][1]からです。\n\n[1]: https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data","metadata":{"papermill":{"duration":0.005196,"end_time":"2023-02-07T01:00:59.092865","exception":false,"start_time":"2023-02-07T01:00:59.087669","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:06:26.337922Z","iopub.execute_input":"2023-02-12T12:06:26.338484Z","iopub.status.idle":"2023-02-12T12:06:26.359772Z","shell.execute_reply.started":"2023-02-12T12:06:26.338452Z","shell.execute_reply":"2023-02-12T12:06:26.358806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nCATS0 = ['text', 'room_fqid','fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# こちらから　https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"papermill":{"duration":0.014685,"end_time":"2023-02-07T01:00:59.112856","exception":false,"start_time":"2023-02-07T01:00:59.098171","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:06:30.371851Z","iopub.execute_input":"2023-02-12T12:06:30.372207Z","iopub.status.idle":"2023-02-12T12:06:30.378327Z","shell.execute_reply.started":"2023-02-12T12:06:30.372174Z","shell.execute_reply":"2023-02-12T12:06:30.377259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATS0:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"papermill":{"duration":0.017716,"end_time":"2023-02-07T01:00:59.136021","exception":false,"start_time":"2023-02-07T01:00:59.118305","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:06:30.379878Z","iopub.execute_input":"2023-02-12T12:06:30.380303Z","iopub.status.idle":"2023-02-12T12:06:30.394058Z","shell.execute_reply.started":"2023-02-12T12:06:30.380265Z","shell.execute_reply":"2023-02-12T12:06:30.392891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:06:30.395636Z","iopub.execute_input":"2023-02-12T12:06:30.395954Z","iopub.status.idle":"2023-02-12T12:08:07.657433Z","shell.execute_reply.started":"2023-02-12T12:06:30.395926Z","shell.execute_reply":"2023-02-12T12:08:07.65634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 以下のカラムを追加した　'mean','max','min','std','sum'\ndf_target_agg_tn = df.groupby(['session_id'])[('text_nunique')].agg(['mean','max','min','std','sum'])\ndf_target_agg_tn.columns = df_target_agg_tn.columns+\"_tn\"\ndf_target_agg_tn_col  = df_target_agg_tn.columns.to_list()\n\ndf_target_agg_rn = df.groupby(['session_id'])[('room_fqid_nunique')].agg(['mean','max','min','std','sum'])\ndf_target_agg_rn.columns = df_target_agg_rn.columns+\"_rn\"\ndf_target_agg_rn_col  = df_target_agg_rn.columns.to_list()\n\ndf_target_agg_fn = df.groupby(['session_id'])[('fqid_nunique')].agg(['mean','max','min','std','sum'])\ndf_target_agg_fn.columns = df_target_agg_fn.columns+\"_fn\"\ndf_target_agg_fn_col  = df_target_agg_fn.columns.to_list()\n\ndf_target_agg_em = df.groupby(['session_id'])[('elapsed_time_mean')].agg(['mean','max','min','std','sum'])\ndf_target_agg_em.columns = df_target_agg_em.columns+\"_em\"\ndf_target_agg_em_col  = df_target_agg_em.columns.to_list()\n\ndf_target_agg_lm = df.groupby(['session_id'])[('level_mean')].agg(['mean','max','min','std','sum'])\ndf_target_agg_lm.columns = df_target_agg_lm.columns+\"_lm\"\ndf_target_agg_lm_col  = df_target_agg_lm.columns.to_list()\n\ndf_target_agg_pm = df.groupby(['session_id'])[('page_mean')].agg(['mean','max','min','std','sum'])\ndf_target_agg_pm.columns = df_target_agg_pm.columns+\"_pm\"\ndf_target_agg_pm_col  = df_target_agg_pm.columns.to_list()\n\ndf_target_agg_hm = df.groupby(['session_id'])[('hover_duration_mean')].agg(['mean','max','min','std','sum'])\ndf_target_agg_hm.columns = df_target_agg_hm.columns+\"_hm\"\ndf_target_agg_hm_col  = df_target_agg_hm.columns.to_list()\n\ndf = pd.merge(df,df_target_agg_tn,left_index=True, right_index=True)\ndf = pd.merge(df,df_target_agg_rn,left_index=True, right_index=True)\ndf = pd.merge(df,df_target_agg_fn,left_index=True, right_index=True)\ndf = pd.merge(df,df_target_agg_em,left_index=True, right_index=True)\ndf = pd.merge(df,df_target_agg_lm,left_index=True, right_index=True)\ndf = pd.merge(df,df_target_agg_pm,left_index=True, right_index=True)\ndf = pd.merge(df,df_target_agg_hm,left_index=True, right_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:13:45.990853Z","iopub.status.idle":"2023-02-12T12:13:45.991263Z","shell.execute_reply.started":"2023-02-12T12:13:45.991044Z","shell.execute_reply":"2023-02-12T12:13:45.991063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_mem = df.memory_usage().sum() / 1024**2\nprint(f'Memory usage of final dataframe is {final_mem:.2f} MB')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Catboost Model\n18の質問に対して、それぞれ1つのモデルを学習させる。さらに、質問1〜3のモデルを学習するためにlevel_groups = '0-4'のデータを使用し、質問4〜13を学習するためにレベルグループ '5-12'、質問14〜18を学習するためにレベルグループ '13-22'を使用しています。なぜなら、これはテスト推論時にKaggleの推論APIから（対応する質問を予測するために）取得するデータだからです。ユーザーの以前のレベルグループのデータを保存し、それを将来のレベルグループの予測に使用することで、モデルを改善することができます。","metadata":{"papermill":{"duration":0.00565,"end_time":"2023-02-07T01:01:33.669525","exception":false,"start_time":"2023-02-07T01:01:33.663875","status":"completed"},"tags":[]}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"papermill":{"duration":0.014699,"end_time":"2023-02-07T01:01:33.689953","exception":false,"start_time":"2023-02-07T01:01:33.675254","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:08:07.818377Z","iopub.execute_input":"2023-02-12T12:08:07.818648Z","iopub.status.idle":"2023-02-12T12:08:07.837375Z","shell.execute_reply.started":"2023-02-12T12:08:07.818623Z","shell.execute_reply":"2023-02-12T12:08:07.836335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# 5 グループ k フォールドで cv スコアを計算する。\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    cat_params = {\n    'loss_function' : 'Logloss',\n    'eval_metric' : 'Logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'num_boost_round': 1000,\n    'od_type' : 'Iter',\n    'od_wait': 50,\n    }\n    \n    # 質問1～18を繰り返す\n    for t in range(1,19):\n        \n        # このTRAIN DATAを使って、こんな質問をする\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        train_pool = Pool(train_x[FEATURES],\n                          train_y['correct'])  \n        test_pool = Pool(valid_x[FEATURES],\n                         valid_y['correct'])\n        clf =  CatBoostClassifier(**cat_params)\n        clf.fit(train_pool, eval_set=[test_pool],verbose=50)\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES])[:,1]\n        \n    print()","metadata":{"papermill":{"duration":69.877213,"end_time":"2023-02-07T01:02:43.57299","exception":false,"start_time":"2023-02-07T01:01:33.695777","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-02-12T12:08:07.907632Z","iopub.execute_input":"2023-02-12T12:08:07.907941Z","iopub.status.idle":"2023-02-12T12:13:45.974077Z","shell.execute_reply.started":"2023-02-12T12:08:07.907913Z","shell.execute_reply":"2023-02-12T12:13:45.972637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(['level_group'],axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 変数重要度を可視化する\n# feature importance のプロット\nplt.figure(figsize=(20, 40))\nsns.set(font_scale = 2.0)\nimportances = pd.Series(clf.feature_importances_, index = df.columns)\nimportances = importances.sort_values()\nimportances.plot(kind = \"barh\")\nplt.title(\"imporance in the catboost \")\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importances.head(15)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute CV Score\n予測確率を `1s` と `0s` に変換する必要がある。競争指標はF1 Scoreであり、これは精度とリコールの調和平均である。F1 Score を最大化するために、`p > threshold` で `1` を予測するときと `0` を予測するときの最適な閾値を求めることにしよう。","metadata":{"papermill":{"duration":0.011241,"end_time":"2023-02-07T01:02:43.59638","exception":false,"start_time":"2023-02-07T01:02:43.585139","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# 18列のデータフレームに真のラベルを入れる\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:13:45.975149Z","iopub.status.idle":"2023-02-12T12:13:45.976353Z","shell.execute_reply.started":"2023-02-12T12:13:45.976108Z","shell.execute_reply":"2023-02-12T12:13:45.976133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# プロブを1と0に変換するための最適なスルーホールドを探す\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:13:45.977513Z","iopub.status.idle":"2023-02-12T12:13:45.978369Z","shell.execute_reply.started":"2023-02-12T12:13:45.978138Z","shell.execute_reply":"2023-02-12T12:13:45.978161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-12T12:13:45.979718Z","iopub.status.idle":"2023-02-12T12:13:45.980092Z","shell.execute_reply.started":"2023-02-12T12:13:45.979902Z","shell.execute_reply":"2023-02-12T12:13:45.979919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"papermill":{"duration":0.771134,"end_time":"2023-02-07T01:02:44.378465","exception":false,"start_time":"2023-02-07T01:02:43.607331","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:13:45.981597Z","iopub.status.idle":"2023-02-12T12:13:45.981982Z","shell.execute_reply.started":"2023-02-12T12:13:45.981797Z","shell.execute_reply":"2023-02-12T12:13:45.981816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test Data","metadata":{"papermill":{"duration":0.011075,"end_time":"2023-02-07T01:02:44.400918","exception":false,"start_time":"2023-02-07T01:02:44.389843","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel train, targets, df, oof, true\n_ = gc.collect()","metadata":{"papermill":{"duration":0.052132,"end_time":"2023-02-07T01:02:44.464739","exception":false,"start_time":"2023-02-07T01:02:44.412607","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:13:45.983259Z","iopub.status.idle":"2023-02-12T12:13:45.983642Z","shell.execute_reply.started":"2023-02-12T12:13:45.983456Z","shell.execute_reply":"2023-02-12T12:13:45.983475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (sample_submission, test) in iter_test:\n    \n    test['page_str'] = test['page'].astype(str)\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # 以下に新たに追加した特徴量\n    df_target_agg_tn = df.groupby(['session_id'])[('text_nunique')].agg(['mean','max','min','std','sum'])\n    df_target_agg_tn.columns = df_target_agg_tn.columns+\"_tn\"\n    df_target_agg_tn_col  = df_target_agg_tn.columns.to_list()\n    \n    df_target_agg_rn = df.groupby(['session_id'])[('room_fqid_nunique')].agg(['mean','max','min','std','sum'])\n    df_target_agg_rn.columns = df_target_agg_rn.columns+\"_rn\"\n    df_target_agg_rn_col  = df_target_agg_rn.columns.to_list()\n    \n    df_target_agg_fn = df.groupby(['session_id'])[('fqid_nunique')].agg(['mean','max','min','std','sum'])\n    df_target_agg_fn.columns = df_target_agg_fn.columns+\"_fn\"\n    df_target_agg_fn_col  = df_target_agg_fn.columns.to_list()\n    \n    df_target_agg_em = df.groupby(['session_id'])[('elapsed_time_mean')].agg(['mean','max','min','std','sum'])\n    df_target_agg_em.columns = df_target_agg_em.columns+\"_em\"\n    df_target_agg_em_col  = df_target_agg_em.columns.to_list()\n    \n    df_target_agg_lm = df.groupby(['session_id'])[('level_mean')].agg(['mean','max','min','std','sum'])\n    df_target_agg_lm.columns = df_target_agg_lm.columns+\"_lm\"\n    df_target_agg_lm_col  = df_target_agg_lm.columns.to_list()\n    \n    df_target_agg_pm = df.groupby(['session_id'])[('page_mean')].agg(['mean','max','min','std','sum'])\n    df_target_agg_pm.columns = df_target_agg_pm.columns+\"_pm\"\n    df_target_agg_pm_col  = df_target_agg_pm.columns.to_list()\n    \n    df_target_agg_hm = df.groupby(['session_id'])[('hover_duration_mean')].agg(['mean','max','min','std','sum'])\n    df_target_agg_hm.columns = df_target_agg_hm.columns+\"_hm\"\n    df_target_agg_hm_col  = df_target_agg_hm.columns.to_list()\n    \n    df = pd.merge(df,df_target_agg_tn,left_index=True, right_index=True)\n    df = pd.merge(df,df_target_agg_rn,left_index=True, right_index=True)\n    df = pd.merge(df,df_target_agg_fn,left_index=True, right_index=True)\n    df = pd.merge(df,df_target_agg_em,left_index=True, right_index=True)\n    df = pd.merge(df,df_target_agg_lm,left_index=True, right_index=True)\n    df = pd.merge(df,df_target_agg_pm,left_index=True, right_index=True)\n    df = pd.merge(df,df_target_agg_hm,left_index=True, right_index=True)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES])[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold)\n    \n    env.predict(sample_submission)","metadata":{"papermill":{"duration":1.002014,"end_time":"2023-02-07T01:02:45.47927","exception":false,"start_time":"2023-02-07T01:02:44.477256","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:13:45.984827Z","iopub.status.idle":"2023-02-12T12:13:45.985635Z","shell.execute_reply.started":"2023-02-12T12:13:45.985427Z","shell.execute_reply":"2023-02-12T12:13:45.985448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.011427,"end_time":"2023-02-07T01:02:45.502331","exception":false,"start_time":"2023-02-07T01:02:45.490904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":0.027432,"end_time":"2023-02-07T01:02:45.541022","exception":false,"start_time":"2023-02-07T01:02:45.51359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:13:45.986828Z","iopub.status.idle":"2023-02-12T12:13:45.987201Z","shell.execute_reply.started":"2023-02-12T12:13:45.98701Z","shell.execute_reply":"2023-02-12T12:13:45.987028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"papermill":{"duration":0.020233,"end_time":"2023-02-07T01:02:45.57314","exception":false,"start_time":"2023-02-07T01:02:45.552907","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-12T12:13:45.988276Z","iopub.status.idle":"2023-02-12T12:13:45.989498Z","shell.execute_reply.started":"2023-02-12T12:13:45.989251Z","shell.execute_reply":"2023-02-12T12:13:45.989275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}