{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n\"\"\"\n#pandasにおけるapplyの高速化のため、pandarallelをinstall\n!pip install pandarallel \n\nimport gc\n\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\nfrom pandarallel import pandarallel\n#pandarallelは初期化が必要\npandarallel.initialize()\n\nBASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\ntrain = pd.read_csv(BASE_DIR / 'train.csv')\n\nnull = np.nan\ntrue = True\nfalse = False\n\nfor col in train.columns:\n\n    if col == 'date': continue\n\n    _index = train[col].notnull()\n    #ネストされているデータを展開\n    train.loc[_index, col] = train.loc[_index, col].parallel_apply(lambda x: eval(x))\n\n    outputs = []\n    #各カラムの値を別データフレームとしてまとめている。\n    for index, date, record in train.loc[_index, ['date', col]].itertuples():\n        _df = pd.DataFrame(record)\n        _df['index'] = index\n        _df['date'] = date\n        outputs.append(_df)\n\n    outputs = pd.concat(outputs).reset_index(drop=True)\n    output_dir = \"../input/MLB_PDEF_train_dataset/\"\n    #各データフレームをcsv, pklとして保存\n    outputs.to_csv(output_dir + '{0}_train.csv'.format(col), index=False)\n    outputs.to_pickle(output_dir + '{0}_train.pkl'.format(col))\n\n    del outputs\n    del train[col]\n    gc.collect()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:33.101243Z","iopub.execute_input":"2021-08-02T01:46:33.101700Z","iopub.status.idle":"2021-08-02T01:46:33.122313Z","shell.execute_reply.started":"2021-08-02T01:46:33.101611Z","shell.execute_reply":"2021-08-02T01:46:33.121184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.metrics import mean_absolute_error\nfrom datetime import timedelta\nfrom functools import reduce\nfrom tqdm import tqdm\nimport lightgbm as lgbm\nimport matplotlib.pyplot as  plt\nimport mlb","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:33.123794Z","iopub.execute_input":"2021-08-02T01:46:33.124066Z","iopub.status.idle":"2021-08-02T01:46:35.452527Z","shell.execute_reply.started":"2021-08-02T01:46:33.124039Z","shell.execute_reply":"2021-08-02T01:46:35.451717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\n#直にデータセットを作っているため、Pathが変な感じ。本来はinputの新しいディレクトリに格納すべき\nTRAIN_DIR = Path('../input/mlb-pdef-train-dataset')","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:35.454008Z","iopub.execute_input":"2021-08-02T01:46:35.454273Z","iopub.status.idle":"2021-08-02T01:46:35.459654Z","shell.execute_reply.started":"2021-08-02T01:46:35.454247Z","shell.execute_reply":"2021-08-02T01:46:35.458647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv(BASE_DIR / 'players.csv')\n\nrosters = pd.read_pickle(TRAIN_DIR / 'rosters_train.pkl')\ntargets = pd.read_pickle(TRAIN_DIR / 'nextDayPlayerEngagement_train.pkl')\nscores = pd.read_pickle(TRAIN_DIR / 'playerBoxScores_train.pkl')\nscores = scores.groupby(['playerId', 'date']).sum().reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:35.461187Z","iopub.execute_input":"2021-08-02T01:46:35.461635Z","iopub.status.idle":"2021-08-02T01:46:40.084728Z","shell.execute_reply.started":"2021-08-02T01:46:35.461594Z","shell.execute_reply":"2021-08-02T01:46:40.083732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets_cols = ['playerId', 'target1', 'target2', 'target3', 'target4', 'date']\nplayers_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status', 'date']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances', 'date']\n\nfeature_cols = ['label_playerId', 'label_primaryPositionName', 'label_teamId',\n       'label_status', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances']","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:40.085997Z","iopub.execute_input":"2021-08-02T01:46:40.086319Z","iopub.status.idle":"2021-08-02T01:46:40.095534Z","shell.execute_reply.started":"2021-08-02T01:46:40.086275Z","shell.execute_reply":"2021-08-02T01:46:40.094419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creat dataset\ntrain = targets[targets_cols].merge(players[players_cols], on=['playerId'], how='left')\ntrain = train.merge(rosters[rosters_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(scores[scores_cols], on=['playerId', 'date'], how='left')","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:40.096523Z","iopub.execute_input":"2021-08-02T01:46:40.096813Z","iopub.status.idle":"2021-08-02T01:46:43.481292Z","shell.execute_reply.started":"2021-08-02T01:46:40.096787Z","shell.execute_reply":"2021-08-02T01:46:43.480293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 内包表記で辞書作成。各IDや名前などをインデックスに紐付ける形で辞書化。\nplayer2num = {c: i for i, c in enumerate(train['playerId'].unique())}\nposition2num = {c: i for i, c in enumerate(train['primaryPositionName'].unique())}\nteamid2num = {c: i for i, c in enumerate(train['teamId'].unique())}\nstatus2num = {c: i for i, c in enumerate(train['status'].unique())}","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:43.482450Z","iopub.execute_input":"2021-08-02T01:46:43.482721Z","iopub.status.idle":"2021-08-02T01:46:43.809634Z","shell.execute_reply.started":"2021-08-02T01:46:43.482693Z","shell.execute_reply":"2021-08-02T01:46:43.808658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#mapと辞書を使って、値の変換を行う。\ntrain['label_playerId'] = train['playerId'].map(player2num)\ntrain['label_primaryPositionName'] = train['primaryPositionName'].map(position2num)\ntrain['label_teamId'] = train['teamId'].map(teamid2num)\ntrain['label_status'] = train['status'].map(status2num)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:43.811361Z","iopub.execute_input":"2021-08-02T01:46:43.811627Z","iopub.status.idle":"2021-08-02T01:46:44.228791Z","shell.execute_reply.started":"2021-08-02T01:46:43.811601Z","shell.execute_reply":"2021-08-02T01:46:44.227775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[feature_cols]\ntrain_y = train[['target1', 'target2', 'target3', 'target4']]","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:44.230021Z","iopub.execute_input":"2021-08-02T01:46:44.230314Z","iopub.status.idle":"2021-08-02T01:46:46.382473Z","shell.execute_reply.started":"2021-08-02T01:46:44.230268Z","shell.execute_reply":"2021-08-02T01:46:46.381616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_index = (train['date'] < 20210401)\n#trainとvalidのデータを用意。※インデックスを振り直しており、元のインデックスは削除。\nx_train = train_X.loc[_index].reset_index(drop=True)\ny_train = train_y.loc[_index].reset_index(drop=True)\nx_valid = train_X.loc[~_index].reset_index(drop=True)\ny_valid = train_y.loc[~_index].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:50.979029Z","iopub.execute_input":"2021-08-02T01:46:50.979379Z","iopub.status.idle":"2021-08-02T01:46:52.605086Z","shell.execute_reply.started":"2021-08-02T01:46:50.979347Z","shell.execute_reply":"2021-08-02T01:46:52.603629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_lgbm(x_train, y_train, x_valid, y_valid, params: dict=None, verbose=100):\n    oof_pred = np.zeros(len(y_valid), dtype=np.float32)\n    model = lgbm.LGBMRegressor(**params)\n    model.fit(x_train, y_train, \n        eval_set=[(x_valid, y_valid)],  \n        early_stopping_rounds=verbose, \n        verbose=verbose)\n    oof_pred = model.predict(x_valid)\n    score = mean_absolute_error(oof_pred, y_valid)\n    print('mae(mean absolute error):', score)\n    return oof_pred, model, score","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:46:59.382064Z","iopub.execute_input":"2021-08-02T01:46:59.382418Z","iopub.status.idle":"2021-08-02T01:46:59.388364Z","shell.execute_reply.started":"2021-08-02T01:46:59.382387Z","shell.execute_reply":"2021-08-02T01:46:59.387161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training lightgbm\nparams = {\n 'objective':'mae',\n 'reg_alpha': 0.1,\n 'reg_lambda': 0.1, \n 'n_estimators': 100000,\n 'learning_rate': 0.1,\n 'random_state': 42,\n}\n\noof1, model1, score1 = fit_lgbm(\n    x_train, y_train['target1'],\n    x_valid, y_valid['target1'],\n    params\n)\noof2, model2, score2 = fit_lgbm(\n    x_train, y_train['target2'],\n    x_valid, y_valid['target2'],\n    params\n)\noof3, model3, score3 = fit_lgbm(\n    x_train, y_train['target3'],\n    x_valid, y_valid['target3'],\n    params\n)\noof4, model4, score4 = fit_lgbm(\n    x_train, y_train['target4'],\n    x_valid, y_valid['target4'],\n    params\n)\n\nscore = (score1+score2+score3+score4) / 4\nprint(f'score: {score}')","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:47:05.939039Z","iopub.execute_input":"2021-08-02T01:47:05.939406Z","iopub.status.idle":"2021-08-02T01:50:19.237034Z","shell.execute_reply.started":"2021-08-02T01:47:05.939375Z","shell.execute_reply":"2021-08-02T01:50:19.236192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度を棒グラフでプロットする関数 \ndef plot_feature_importance(df): \n    n_features = len(df)\n    df_plot = df.sort_values('importance')\n    f_importance_plot = df_plot['importance'].values\n    plt.barh(range(n_features), f_importance_plot, align='center') \n    cols_plot = df_plot['feature'].values\n    plt.yticks(np.arange(n_features), cols_plot)\n    plt.xlabel('Feature importance')\n    plt.ylabel('Feature')","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:50:19.238452Z","iopub.execute_input":"2021-08-02T01:50:19.238933Z","iopub.status.idle":"2021-08-02T01:50:19.244746Z","shell.execute_reply.started":"2021-08-02T01:50:19.238901Z","shell.execute_reply":"2021-08-02T01:50:19.243723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の算出 (データフレームで取得)\ncols = list(train_X.columns)\n# 特徴量重要度の算出\nf_importance1 = np.array(model1.feature_importances_)\nf_importance1 = f_importance1 / np.sum(f_importance1) # 正規化\ndf_importance1 = pd.DataFrame({'feature':cols, 'importance':f_importance1})\ndf_importance1 = df_importance1.sort_values('importance', ascending=False)\ndf_importance1.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:51:37.331611Z","iopub.execute_input":"2021-08-02T01:51:37.331979Z","iopub.status.idle":"2021-08-02T01:51:37.352332Z","shell.execute_reply.started":"2021-08-02T01:51:37.331945Z","shell.execute_reply":"2021-08-02T01:51:37.351377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の可視化\nplt.figure(figsize = (10, 10))\nplot_feature_importance(df_importance1.iloc[:30])","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:51:38.467489Z","iopub.execute_input":"2021-08-02T01:51:38.467824Z","iopub.status.idle":"2021-08-02T01:51:38.905389Z","shell.execute_reply.started":"2021-08-02T01:51:38.467794Z","shell.execute_reply":"2021-08-02T01:51:38.904353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の算出 (データフレームで取得)\ncols = list(train_X.columns)\n# 特徴量重要度の算出\nf_importance2 = np.array(model2.feature_importances_)\nf_importance2 = f_importance2 / np.sum(f_importance2) # 正規化\ndf_importance2 = pd.DataFrame({'feature':cols, 'importance':f_importance2})\ndf_importance2 = df_importance2.sort_values('importance', ascending=False)\ndf_importance2.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:54:00.566427Z","iopub.execute_input":"2021-08-02T01:54:00.566792Z","iopub.status.idle":"2021-08-02T01:54:00.582319Z","shell.execute_reply.started":"2021-08-02T01:54:00.566761Z","shell.execute_reply":"2021-08-02T01:54:00.581114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の可視化\nplt.figure(figsize = (10, 10))\nplot_feature_importance(df_importance2.iloc[:30])","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:54:20.975379Z","iopub.execute_input":"2021-08-02T01:54:20.975748Z","iopub.status.idle":"2021-08-02T01:54:21.336215Z","shell.execute_reply.started":"2021-08-02T01:54:20.975717Z","shell.execute_reply":"2021-08-02T01:54:21.335246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の算出 (データフレームで取得)\ncols = list(train_X.columns)\n# 特徴量重要度の算出\nf_importance3 = np.array(model3.feature_importances_)\nf_importance3 = f_importance3 / np.sum(f_importance3) # 正規化\ndf_importance3 = pd.DataFrame({'feature':cols, 'importance':f_importance3})\ndf_importance3 = df_importance3.sort_values('importance', ascending=False)\ndf_importance3.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:55:17.654042Z","iopub.execute_input":"2021-08-02T01:55:17.654425Z","iopub.status.idle":"2021-08-02T01:55:17.668903Z","shell.execute_reply.started":"2021-08-02T01:55:17.654390Z","shell.execute_reply":"2021-08-02T01:55:17.667937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の可視化\nplt.figure(figsize = (10, 10))\nplot_feature_importance(df_importance3.iloc[:30])","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:55:18.324694Z","iopub.execute_input":"2021-08-02T01:55:18.325061Z","iopub.status.idle":"2021-08-02T01:55:18.712586Z","shell.execute_reply.started":"2021-08-02T01:55:18.325018Z","shell.execute_reply":"2021-08-02T01:55:18.711418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の算出 (データフレームで取得)\ncols = list(train_X.columns)\n# 特徴量重要度の算出\nf_importance4 = np.array(model4.feature_importances_)\nf_importance4 = f_importance4 / np.sum(f_importance4) # 正規化\ndf_importance4 = pd.DataFrame({'feature':cols, 'importance':f_importance4})\ndf_importance4 = df_importance4.sort_values('importance', ascending=False)\ndf_importance4.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:55:27.499002Z","iopub.execute_input":"2021-08-02T01:55:27.499363Z","iopub.status.idle":"2021-08-02T01:55:27.513074Z","shell.execute_reply.started":"2021-08-02T01:55:27.499331Z","shell.execute_reply":"2021-08-02T01:55:27.512030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量重要度の可視化\nplt.figure(figsize = (10, 10))\nplot_feature_importance(df_importance4.iloc[:30])","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:55:28.646221Z","iopub.execute_input":"2021-08-02T01:55:28.646713Z","iopub.status.idle":"2021-08-02T01:55:28.993246Z","shell.execute_reply.started":"2021-08-02T01:55:28.646679Z","shell.execute_reply":"2021-08-02T01:55:28.992273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n#ハイパーパラメータのチューニング用。時間がかなりかかるため、自環境で実施し、パラメータの設定だけ、kaggle上に反映して、実行した方が良さそう。\nimport optuna.integration.lightgbm as op_lgbm \ndef op_fit_lgbm(x_train, y_train, x_valid, y_valid, params: dict=None, verbose=100):\n    oof_pred = np.zeros(len(y_valid), dtype=np.float32) \n    trains = op_lgbm.Dataset(x_train, y_train) \n    valids = op_lgbm.Dataset(x_valid, y_valid) \n    model = op_lgbm.train( params, trains, valid_sets = valids, num_boost_round = 10000, verbose_eval = False, early_stopping_rounds = 100 ) \n    best_params = model.params \n    oof_pred = model.predict(x_valid) \n    score = mean_absolute_error(oof_pred, y_valid) \n    print('mae:', score) \n    return oof_pred, model, score, best_params \n    \nparams = { 'objective':'mae', 'metric':'mae' } \noof1, model1, score1, best_params1 = op_fit_lgbm( x_train, y_train['target1'], x_valid, y_valid['target1'], params ) \noof2, model2, score2, best_params2 = op_fit_lgbm( x_train, y_train['target2'], x_valid, y_valid['target2'], params ) \noof3, model3, score3, best_params3 = op_fit_lgbm( x_train, y_train['target3'], x_valid, y_valid['target3'], params ) \noof4, model4, score4, best_params4 = op_fit_lgbm( x_train, y_train['target4'], x_valid, y_valid['target4'], params ) \nprint(best_params1)\nprint(best_params2)\nprint(best_params3)\nprint(best_params4)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:57:06.956380Z","iopub.execute_input":"2021-08-02T01:57:06.956748Z","iopub.status.idle":"2021-08-02T01:57:06.963212Z","shell.execute_reply.started":"2021-08-02T01:57:06.956714Z","shell.execute_reply":"2021-08-02T01:57:06.962182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances']\n\nnull = np.nan\ntrue = True\nfalse = False\n\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in iter_test: # make predictions here\n    \n    sample_prediction_df = sample_prediction_df.reset_index(drop=True)\n    \n    # creat dataset\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    # Dealing with missing values\n    if test_df['rosters'].iloc[0] == test_df['rosters'].iloc[0]:\n        test_rosters = pd.DataFrame(eval(test_df['rosters'].iloc[0]))\n    else:\n        test_rosters = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in rosters.columns:\n            if col == 'playerId': continue\n            test_rosters[col] = np.nan\n            \n    if test_df['playerBoxScores'].iloc[0] == test_df['playerBoxScores'].iloc[0]:\n        test_scores = pd.DataFrame(eval(test_df['playerBoxScores'].iloc[0]))\n    else:\n        test_scores = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in scores.columns:\n            if col == 'playerId': continue\n            test_scores[col] = np.nan\n    test_scores = test_scores.groupby('playerId').sum().reset_index()\n    test = sample_prediction_df[['playerId']].copy()\n    test = test.merge(players[players_cols], on='playerId', how='left')\n    test = test.merge(test_rosters[rosters_cols], on='playerId', how='left')\n    test = test.merge(test_scores[scores_cols], on='playerId', how='left')\n\n    test['label_playerId'] = test['playerId'].map(player2num)\n    test['label_primaryPositionName'] = test['primaryPositionName'].map(position2num)\n    test['label_teamId'] = test['teamId'].map(teamid2num)\n    test['label_status'] = test['status'].map(status2num)\n    \n    test_X = test[feature_cols]\n    \n    # predict\n    pred1 = model1.predict(test_X)\n    pred2 = model2.predict(test_X)\n    pred3 = model3.predict(test_X)\n    pred4 = model4.predict(test_X)\n    \n    # merge submission\n    sample_prediction_df['target1'] = np.clip(pred1, 0, 100)\n    sample_prediction_df['target2'] = np.clip(pred2, 0, 100)\n    sample_prediction_df['target3'] = np.clip(pred3, 0, 100)\n    sample_prediction_df['target4'] = np.clip(pred4, 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)\n    del sample_prediction_df['playerId']\n    \n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T01:57:37.719612Z","iopub.execute_input":"2021-08-02T01:57:37.719958Z","iopub.status.idle":"2021-08-02T01:57:39.769640Z","shell.execute_reply.started":"2021-08-02T01:57:37.719926Z","shell.execute_reply":"2021-08-02T01:57:39.768561Z"},"trusted":true},"execution_count":null,"outputs":[]}]}