{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\n# from pandarallel import pandarallel\n# pandarallel.initialize()\n\nBASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting/')\n#print(-1)\ntrain = pd.read_csv(BASE_DIR / 'train_updated.csv')\n#print(0)\nnull = np.nan\ntrue = True\nfalse = False\n\nfor col in [\"rosters\", \"nextDayPlayerEngagement\", \"playerBoxScores\"]:\n\n    if col == 'date': continue\n\n    _index = train[col].notnull()\n    train.loc[_index, col] = train.loc[_index, col].apply(lambda x: eval(x))\n\n    outputs = []\n    #print(1)\n    for index, date, record in train.loc[_index, ['date', col]].itertuples():\n        _df = pd.DataFrame(record)\n        _df['index'] = index\n        _df['date'] = date\n        outputs.append(_df)\n    #print(2)\n    outputs = pd.concat(outputs).reset_index(drop=True)\n    if col == \"rosters\":\n        rosters = outputs.copy()\n    elif col == \"nextDayPlayerEngagement\":\n        targets = outputs.copy()\n    elif col == \"playerBoxScores\":\n        scores = outputs.copy()\n    \n\n    del outputs, _index, _df\n    del train[col]\n    gc.collect()","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-07-31T18:47:28.958314Z","iopub.status.busy":"2021-07-31T18:47:28.957549Z","iopub.status.idle":"2021-07-31T18:52:24.490588Z","shell.execute_reply":"2021-07-31T18:52:24.49157Z","shell.execute_reply.started":"2021-07-31T17:30:43.033737Z"},"papermill":{"duration":295.593245,"end_time":"2021-07-31T18:52:24.492538","exception":false,"start_time":"2021-07-31T18:47:28.899293","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2  \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)  \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:24.57665Z","iopub.status.busy":"2021-07-31T18:52:24.57595Z","iopub.status.idle":"2021-07-31T18:52:24.579108Z","shell.execute_reply":"2021-07-31T18:52:24.578612Z","shell.execute_reply.started":"2021-07-31T17:35:29.958785Z"},"papermill":{"duration":0.051846,"end_time":"2021-07-31T18:52:24.579246","exception":false,"start_time":"2021-07-31T18:52:24.5274","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv(BASE_DIR / 'players.csv')\nscores['positionCode'] = scores['positionCode'].astype(int)\nscores = scores.groupby(['playerId', 'date']).sum().reset_index()","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:24.655714Z","iopub.status.busy":"2021-07-31T18:52:24.655041Z","iopub.status.idle":"2021-07-31T18:52:25.617627Z","shell.execute_reply":"2021-07-31T18:52:25.61688Z","shell.execute_reply.started":"2021-07-31T17:35:29.977187Z"},"papermill":{"duration":1.001998,"end_time":"2021-07-31T18:52:25.617797","exception":false,"start_time":"2021-07-31T18:52:24.615799","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.metrics import mean_absolute_error\nfrom datetime import timedelta\nfrom functools import reduce\nfrom tqdm import tqdm\nimport lightgbm as lgbm\n\nimport mlb","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:25.690796Z","iopub.status.busy":"2021-07-31T18:52:25.69009Z","iopub.status.idle":"2021-07-31T18:52:27.791986Z","shell.execute_reply":"2021-07-31T18:52:27.792501Z","shell.execute_reply.started":"2021-07-31T17:35:31.056476Z"},"papermill":{"duration":2.140931,"end_time":"2021-07-31T18:52:27.792717","exception":false,"start_time":"2021-07-31T18:52:25.651786","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets_cols = ['playerId', 'target1', 'target2', 'target3', 'target4', 'date']\nplayers_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status', 'date']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances', 'date', 'positionCode']\n\nfeature_cols = [\n    'label_playerId', 'label_primaryPositionName', 'label_teamId',\n       'label_status',\n    'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances','target1_mean',\n 'target1_median',\n 'target1_std',\n 'target1_min',\n 'target1_max',\n 'target1_prob',\n 'target2_mean',\n 'target2_median',\n 'target2_std',\n 'target2_min',\n 'target2_max',\n 'target2_prob',\n 'target3_mean',\n 'target3_median',\n 'target3_std',\n 'target3_min',\n 'target3_max',\n 'target3_prob',\n 'target4_mean',\n 'target4_median',\n 'target4_std',\n 'target4_min',\n 'target4_max',\n 'target4_prob',\n    'positionCode'\n]\n","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:27.876463Z","iopub.status.busy":"2021-07-31T18:52:27.875812Z","iopub.status.idle":"2021-07-31T18:52:27.87909Z","shell.execute_reply":"2021-07-31T18:52:27.87858Z","shell.execute_reply.started":"2021-07-31T17:35:48.824334Z"},"papermill":{"duration":0.051752,"end_time":"2021-07-31T18:52:27.879227","exception":false,"start_time":"2021-07-31T18:52:27.827475","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_target_stats = pd.read_csv(\"../input/player-target-stats/player_target_stats.csv\")\ndata_names=player_target_stats.columns.values.tolist()","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:27.952363Z","iopub.status.busy":"2021-07-31T18:52:27.951708Z","iopub.status.idle":"2021-07-31T18:52:27.991471Z","shell.execute_reply":"2021-07-31T18:52:27.990937Z","shell.execute_reply.started":"2021-07-31T17:36:00.908975Z"},"papermill":{"duration":0.078351,"end_time":"2021-07-31T18:52:27.991619","exception":false,"start_time":"2021-07-31T18:52:27.913268","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creat dataset\ntrain = targets[targets_cols].merge(players[players_cols], on=['playerId'], how='left')\ntrain = train.merge(rosters[rosters_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(scores[scores_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(player_target_stats, how='left', left_on=[\"playerId\"],right_on=[\"playerId\"])","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:28.067036Z","iopub.status.busy":"2021-07-31T18:52:28.066358Z","iopub.status.idle":"2021-07-31T18:52:34.085047Z","shell.execute_reply":"2021-07-31T18:52:34.085729Z","shell.execute_reply.started":"2021-07-31T17:36:08.067135Z"},"papermill":{"duration":6.060173,"end_time":"2021-07-31T18:52:34.085945","exception":false,"start_time":"2021-07-31T18:52:28.025772","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label encoding\nplayer2num = {c: i for i, c in enumerate(train['playerId'].unique())}\nposition2num = {c: i for i, c in enumerate(train['primaryPositionName'].unique())}\nteamid2num = {c: i for i, c in enumerate(train['teamId'].unique())}\nstatus2num = {c: i for i, c in enumerate(train['status'].unique())}\ntrain['label_playerId'] = train['playerId'].map(player2num)\ntrain['label_primaryPositionName'] = train['primaryPositionName'].map(position2num)\ntrain['label_teamId'] = train['teamId'].map(teamid2num)\ntrain['label_status'] = train['status'].map(status2num)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:34.163938Z","iopub.status.busy":"2021-07-31T18:52:34.162933Z","iopub.status.idle":"2021-07-31T18:52:35.376888Z","shell.execute_reply":"2021-07-31T18:52:35.376243Z","shell.execute_reply.started":"2021-07-31T17:36:14.405274Z"},"papermill":{"duration":1.256349,"end_time":"2021-07-31T18:52:35.377041","exception":false,"start_time":"2021-07-31T18:52:34.120692","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seasons_df = pd.read_csv(BASE_DIR / 'seasons.csv')\nseasons_df['seasonStartDate'] = pd.to_datetime(seasons_df['seasonStartDate'], format='%Y-%m-%d')\nall_cols = train.columns.values.tolist()\ntrain['year'] = pd.to_datetime(train['date'], format = '%Y%m%d').dt.year\ntrain = pd.merge(train,\n                   seasons_df,\n                   how = 'left',\n                   left_on = 'year',\n                   right_on = 'seasonId')\ntrain['days_to_season_start'] = (train['seasonStartDate'] - pd.to_datetime(train['date'], format = '%Y%m%d')).dt.days\nall_cols.append('days_to_season_start')\nfeature_cols.append('days_to_season_start')\ntrain = train[all_cols]","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:35.453174Z","iopub.status.busy":"2021-07-31T18:52:35.452476Z","iopub.status.idle":"2021-07-31T18:52:41.926599Z","shell.execute_reply":"2021-07-31T18:52:41.92597Z","shell.execute_reply.started":"2021-07-31T17:36:18.741507Z"},"papermill":{"duration":6.515376,"end_time":"2021-07-31T18:52:41.926768","exception":false,"start_time":"2021-07-31T18:52:35.411392","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"isGame\"] = (train.positionCode.notna()).astype(np.int32)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:42.000861Z","iopub.status.busy":"2021-07-31T18:52:42.000202Z","iopub.status.idle":"2021-07-31T18:52:42.012095Z","shell.execute_reply":"2021-07-31T18:52:42.011516Z","shell.execute_reply.started":"2021-07-31T17:37:40.438674Z"},"papermill":{"duration":0.051346,"end_time":"2021-07-31T18:52:42.012243","exception":false,"start_time":"2021-07-31T18:52:41.960897","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ntrain[\"daysAfterGame\"] = np.nan\ntrain[\"totalGames\"] = np.nan\n\nfor player in tqdm(train.playerId.unique()):\n    player_mask = train.playerId == player\n    result = []\n    result_count = []\n    value = 0\n    count = 0\n    for i in train.loc[player_mask, \"isGame\"]:\n        if i == 1:\n            count += 1\n            result.append(0)\n            result_count.append(count)\n            value = 0\n        else:\n            value += 1\n            result.append(value)\n            result_count.append(count)\n    train.loc[player_mask, \"daysAfterGame\"] = result\n    train.loc[player_mask, \"totalGames\"] = result_count","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:52:42.088186Z","iopub.status.busy":"2021-07-31T18:52:42.087498Z","iopub.status.idle":"2021-07-31T18:54:03.591837Z","shell.execute_reply":"2021-07-31T18:54:03.590926Z","shell.execute_reply.started":"2021-07-31T17:37:47.109287Z"},"papermill":{"duration":81.545863,"end_time":"2021-07-31T18:54:03.592089","exception":false,"start_time":"2021-07-31T18:52:42.046226","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['year'] = pd.to_datetime(train['date'], format = '%Y%m%d').dt.year\ntrain['month'] = pd.to_datetime(train['date'], format = '%Y%m%d').dt.month","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:04.078172Z","iopub.status.busy":"2021-07-31T18:54:04.077506Z","iopub.status.idle":"2021-07-31T18:54:04.654383Z","shell.execute_reply":"2021-07-31T18:54:04.654876Z","shell.execute_reply.started":"2021-07-31T17:39:06.927156Z"},"papermill":{"duration":0.821695,"end_time":"2021-07-31T18:54:04.655064","exception":false,"start_time":"2021-07-31T18:54:03.833369","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = []\nfor start, end in zip(seasons_df.seasonStartDate.astype(str)[1:], seasons_df.seasonEndDate.astype(str)[1:]):\n    start = int(start.replace(\"-\", \"\"))\n    end = int(end.replace(\"-\", \"\"))\n    \n    mask = np.logical_and(\n        train.date >= start,\n        train.date <= end\n    )\n    \n    dfs.append(\n        train[mask]\n    \n    )\ntrain = pd.concat(dfs, axis=0)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:05.161484Z","iopub.status.busy":"2021-07-31T18:54:05.160792Z","iopub.status.idle":"2021-07-31T18:54:08.529711Z","shell.execute_reply":"2021-07-31T18:54:08.529036Z","shell.execute_reply.started":"2021-07-31T17:39:07.51834Z"},"papermill":{"duration":3.621214,"end_time":"2021-07-31T18:54:08.529869","exception":false,"start_time":"2021-07-31T18:54:04.908655","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols += [\"daysAfterGame\", 'isGame']\n","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:09.012052Z","iopub.status.busy":"2021-07-31T18:54:09.011263Z","iopub.status.idle":"2021-07-31T18:54:09.013322Z","shell.execute_reply":"2021-07-31T18:54:09.013993Z","shell.execute_reply.started":"2021-07-31T17:39:11.170892Z"},"papermill":{"duration":0.245137,"end_time":"2021-07-31T18:54:09.014184","exception":false,"start_time":"2021-07-31T18:54:08.769047","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"rankMonth\"] = train.groupby(\"year\").month.transform(\"rank\", method=\"dense\")\ntrain_stats = train[train.rankMonth.isin([3, 4, 5, 6])].reset_index(drop=True)\n\nstats = train_stats.groupby([\"playerId\", \"year\"])[[\n        \"target1\", \"target2\", \"target3\", \"target4\"\n    ]].agg(\n    [\"mean\", \"std\", \"median\", \"max\", \"min\"]\n)\n\n\nstats.columns = [f\"3year_{c[1]}_{c[0]}\"for c in stats.columns]\ntrain = train.merge(stats.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:09.494285Z","iopub.status.busy":"2021-07-31T18:54:09.493541Z","iopub.status.idle":"2021-07-31T18:54:15.212283Z","shell.execute_reply":"2021-07-31T18:54:15.211726Z","shell.execute_reply.started":"2021-07-31T17:45:21.88106Z"},"papermill":{"duration":5.959773,"end_time":"2021-07-31T18:54:15.212433","exception":false,"start_time":"2021-07-31T18:54:09.25266","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols += stats.columns.tolist()","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:15.693603Z","iopub.status.busy":"2021-07-31T18:54:15.692978Z","iopub.status.idle":"2021-07-31T18:54:15.694961Z","shell.execute_reply":"2021-07-31T18:54:15.695388Z","shell.execute_reply.started":"2021-07-31T17:45:28.813943Z"},"papermill":{"duration":0.243953,"end_time":"2021-07-31T18:54:15.695564","exception":false,"start_time":"2021-07-31T18:54:15.451611","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"rankMonth\"] = train.groupby(\"year\").month.transform(\"rank\", method=\"dense\")\ntrain_stats = train[train.rankMonth.isin([3, 4, 5, 6])].reset_index(drop=True)\n\nstats_team = train_stats.groupby([\"teamId\", \"year\"])[[\n        \"target1\", \"target2\", \"target3\", \"target4\"\n    ]].agg(\n    [\"mean\", \"std\", \"median\", \"max\", \"min\"]\n)\n\n\nstats_team.columns = [f\"3year_{c[1]}_{c[0]}_teamId\" for c in stats_team.columns]\ntrain = train.merge(stats_team.reset_index(), how=\"left\", on=[\"teamId\", \"year\"])","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:16.203594Z","iopub.status.busy":"2021-07-31T18:54:16.202699Z","iopub.status.idle":"2021-07-31T18:54:22.985257Z","shell.execute_reply":"2021-07-31T18:54:22.984645Z","shell.execute_reply.started":"2021-07-31T17:45:28.821059Z"},"papermill":{"duration":7.028702,"end_time":"2021-07-31T18:54:22.985418","exception":false,"start_time":"2021-07-31T18:54:15.956716","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols += stats_team.columns.tolist()\n","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:23.467944Z","iopub.status.busy":"2021-07-31T18:54:23.467278Z","iopub.status.idle":"2021-07-31T18:54:23.470599Z","shell.execute_reply":"2021-07-31T18:54:23.47013Z","shell.execute_reply.started":"2021-07-31T17:45:35.81074Z"},"papermill":{"duration":0.246009,"end_time":"2021-07-31T18:54:23.470736","exception":false,"start_time":"2021-07-31T18:54:23.224727","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"rankMonth\"] = train.groupby(\"year\").month.transform(\"rank\", method=\"dense\")\ntrain[\"shiftIsGame\"] = train.groupby(\"playerId\").isGame.transform(\"shift\", periods=-1)\ntrain_stats = train[train.rankMonth.isin([3, 4, 5, 6])].reset_index(drop=True)\ntrain_stats = train_stats[train_stats.shiftIsGame == 1].reset_index(drop=True)\n\n\nstats_code = train_stats.groupby([\"year\", \"playerId\"])[[\n        \"target1\", \"target2\", \"target3\", \"target4\"\n    ]].agg(\n    [\"mean\", \"std\", \"median\", \"max\", \"min\"]\n)\n\n\nstats_code.columns = [f\"3year_{c[1]}_{c[0]}_shiftIsGame1\" for c in stats_code.columns]\ntrain = train.merge(stats_code.reset_index(), how=\"left\", on=[\"year\", \"playerId\"])","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:23.951967Z","iopub.status.busy":"2021-07-31T18:54:23.951238Z","iopub.status.idle":"2021-07-31T18:54:31.447772Z","shell.execute_reply":"2021-07-31T18:54:31.447105Z","shell.execute_reply.started":"2021-07-31T17:45:35.817128Z"},"papermill":{"duration":7.740768,"end_time":"2021-07-31T18:54:31.447917","exception":false,"start_time":"2021-07-31T18:54:23.707149","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols += stats_code.columns.tolist()\n","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:31.934816Z","iopub.status.busy":"2021-07-31T18:54:31.934099Z","iopub.status.idle":"2021-07-31T18:54:31.93665Z","shell.execute_reply":"2021-07-31T18:54:31.937119Z","shell.execute_reply.started":"2021-07-31T17:45:43.652425Z"},"papermill":{"duration":0.250122,"end_time":"2021-07-31T18:54:31.937297","exception":false,"start_time":"2021-07-31T18:54:31.687175","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"rankMonth\"] = train.groupby(\"year\").month.transform(\"rank\", method=\"dense\")\ntrain_stats = train[train.rankMonth.isin([6])].reset_index(drop=True)\n\nstats1 = train_stats.groupby([\"playerId\", \"year\"])[[\n        \"target1\", \"target2\", \"target3\", \"target4\"\n    ]].agg(\n    [\"mean\", \"std\", \"median\", \"max\", \"min\"]\n)\n\n\nstats1.columns = [f\"3year_{c[1]}_{c[0]}_1\"for c in stats1.columns]\ntrain = train.merge(stats1.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:32.537851Z","iopub.status.busy":"2021-07-31T18:54:32.537159Z","iopub.status.idle":"2021-07-31T18:54:39.737569Z","shell.execute_reply":"2021-07-31T18:54:39.736982Z","shell.execute_reply.started":"2021-07-31T17:45:43.658846Z"},"papermill":{"duration":7.448418,"end_time":"2021-07-31T18:54:39.737707","exception":false,"start_time":"2021-07-31T18:54:32.289289","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols += stats1.columns.tolist()\nfeature_cols += [\"year\", \"rankMonth\"]","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:40.222173Z","iopub.status.busy":"2021-07-31T18:54:40.221474Z","iopub.status.idle":"2021-07-31T18:54:40.224104Z","shell.execute_reply":"2021-07-31T18:54:40.224674Z","shell.execute_reply.started":"2021-07-31T17:45:50.608307Z"},"papermill":{"duration":0.245549,"end_time":"2021-07-31T18:54:40.22486","exception":false,"start_time":"2021-07-31T18:54:39.979311","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"rankMonth\"] = train.groupby(\"year\").month.transform(\"rank\", method=\"dense\")\ntrain_stats = train[train.rankMonth.isin([5, 6])].reset_index(drop=True)\n\nstats2 = train_stats.groupby([\"playerId\", \"year\"])[[\n        \"target1\", \"target2\", \"target3\", \"target4\"\n    ]].agg(\n    [\"mean\", \"std\", \"median\", \"max\", \"min\"]\n)\n\n\nstats2.columns = [f\"3year_{c[1]}_{c[0]}_2\" for c in stats2.columns]\ntrain = train.merge(stats2.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:40.712025Z","iopub.status.busy":"2021-07-31T18:54:40.711311Z","iopub.status.idle":"2021-07-31T18:54:49.548475Z","shell.execute_reply":"2021-07-31T18:54:49.547889Z"},"papermill":{"duration":9.083468,"end_time":"2021-07-31T18:54:49.548627","exception":false,"start_time":"2021-07-31T18:54:40.465159","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols += stats2.columns.tolist()\n","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:50.032947Z","iopub.status.busy":"2021-07-31T18:54:50.03229Z","iopub.status.idle":"2021-07-31T18:54:50.034224Z","shell.execute_reply":"2021-07-31T18:54:50.03467Z"},"papermill":{"duration":0.245654,"end_time":"2021-07-31T18:54:50.034857","exception":false,"start_time":"2021-07-31T18:54:49.789203","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = train[train.date < 20210501].reset_index(drop=True)  # use all data","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:50.516427Z","iopub.status.busy":"2021-07-31T18:54:50.515769Z","iopub.status.idle":"2021-07-31T18:54:50.518605Z","shell.execute_reply":"2021-07-31T18:54:50.519075Z","shell.execute_reply.started":"2021-07-31T17:45:50.614809Z"},"papermill":{"duration":0.244633,"end_time":"2021-07-31T18:54:50.519251","exception":false,"start_time":"2021-07-31T18:54:50.274618","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[train.playerId.isin(players[players.playerForTestSetAndFuturePreds.fillna(False)].playerId)]","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:50.996476Z","iopub.status.busy":"2021-07-31T18:54:50.995804Z","iopub.status.idle":"2021-07-31T18:54:53.872556Z","shell.execute_reply":"2021-07-31T18:54:53.871591Z","shell.execute_reply.started":"2021-07-31T17:45:50.625892Z"},"papermill":{"duration":3.117105,"end_time":"2021-07-31T18:54:53.872703","exception":false,"start_time":"2021-07-31T18:54:50.755598","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.reset_index(drop=True)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:54.351081Z","iopub.status.busy":"2021-07-31T18:54:54.35038Z","iopub.status.idle":"2021-07-31T18:54:54.888045Z","shell.execute_reply":"2021-07-31T18:54:54.887443Z","shell.execute_reply.started":"2021-07-31T17:45:54.229761Z"},"papermill":{"duration":0.778929,"end_time":"2021-07-31T18:54:54.888223","exception":false,"start_time":"2021-07-31T18:54:54.109294","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def use_reduce_memory_usage():\n    datasets = [train_X, train_y, x_train1, y_train1, x_valid1, y_valid1,\n                x_train2, y_train2, x_valid2, y_valid2]\n\n    for data in datasets:\n        reduce_mem_usage(data)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:55.390832Z","iopub.status.busy":"2021-07-31T18:54:55.390069Z","iopub.status.idle":"2021-07-31T18:54:55.393069Z","shell.execute_reply":"2021-07-31T18:54:55.392485Z","shell.execute_reply.started":"2021-07-31T17:45:54.743106Z"},"papermill":{"duration":0.249274,"end_time":"2021-07-31T18:54:55.393215","exception":false,"start_time":"2021-07-31T18:54:55.143941","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:55.873459Z","iopub.status.busy":"2021-07-31T18:54:55.87242Z","iopub.status.idle":"2021-07-31T18:54:55.876361Z","shell.execute_reply":"2021-07-31T18:54:55.876855Z","shell.execute_reply.started":"2021-07-31T17:45:54.75008Z"},"papermill":{"duration":0.244351,"end_time":"2021-07-31T18:54:55.877035","exception":false,"start_time":"2021-07-31T18:54:55.632684","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_lgbm_(\n    train, features, target, params\n\n):\n    params[\"n_estimators\"] = 5000\n    models = []\n    oof = np.zeros(train.shape[0])\n    cv = GroupKFold(n_splits=4)\n    # for random_state in [2344, 33, 768, 42]:\n    for train_idx, valid_idx in cv.split(train, train, groups=train.year):\n\n        X_train, X_valid = train.loc[train_idx, features], train.loc[valid_idx, features]\n        y_train, y_valid = train.loc[train_idx, target], train.loc[valid_idx, target]\n\n        model = lgbm.LGBMRegressor(**params, n_jobs=-1,\n                                   # random_state=random_state\n                                   )\n\n        model.fit(\n            X_train, y_train,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=100,\n            verbose=100\n        )\n\n        models.append(model)\n        oof[valid_idx] = model.predict(X_valid)\n    return models, oof","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:56.359357Z","iopub.status.busy":"2021-07-31T18:54:56.358689Z","iopub.status.idle":"2021-07-31T18:54:56.366303Z","shell.execute_reply":"2021-07-31T18:54:56.366764Z","shell.execute_reply.started":"2021-07-31T17:45:54.765699Z"},"papermill":{"duration":0.248739,"end_time":"2021-07-31T18:54:56.366936","exception":false,"start_time":"2021-07-31T18:54:56.118197","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols = list(set(feature_cols) - set(['hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances']))","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:56.850392Z","iopub.status.busy":"2021-07-31T18:54:56.849717Z","iopub.status.idle":"2021-07-31T18:54:56.855164Z","shell.execute_reply":"2021-07-31T18:54:56.854582Z","shell.execute_reply.started":"2021-07-31T17:45:54.780871Z"},"papermill":{"duration":0.249613,"end_time":"2021-07-31T18:54:56.855307","exception":false,"start_time":"2021-07-31T18:54:56.605694","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params1 = {'objective':'mae','reg_alpha': 0.14947461820098767, 'reg_lambda': 0.10185644384043743,\n           'learning_rate': 0.08046301304430488, 'num_leaves': 674, 'feature_fraction': 0.9101240539122566,\n           'bagging_fraction': 0.9884451442950513, 'bagging_freq': 8, 'min_child_samples': 51}\n\nmodels1_, oof1_ = fit_lgbm_(\n    train, feature_cols, \"target1\", params1\n)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:54:57.33801Z","iopub.status.busy":"2021-07-31T18:54:57.337369Z","iopub.status.idle":"2021-07-31T19:05:36.778131Z","shell.execute_reply":"2021-07-31T19:05:36.778825Z","shell.execute_reply.started":"2021-07-31T17:45:54.798947Z"},"papermill":{"duration":639.685895,"end_time":"2021-07-31T19:05:36.779088","exception":false,"start_time":"2021-07-31T18:54:57.093193","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params2 = {'bagging_fraction': 0.87, 'boosting': 'gbdt', 'feature_fraction': 0.58, 'lambda_l1': 4.572164206128591,\n          'lambda_l2': 2.5326666967819307, 'learning_rate': 0.03180599834934879, 'max_bin': 170, 'max_depth': 7,\n          'metric': 'mae', 'min_data_in_bin': 43, 'min_data_in_leaf': 205, 'min_gain_to_split': 0.35000000000000003,\n          'num_leaves': 119, 'objective': 'mae', 'subsample': 0.8363057030853551}\n\nmodels2_, oof2_ = fit_lgbm_(\n    train, feature_cols, \"target2\", params2\n)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T19:05:37.290063Z","iopub.status.busy":"2021-07-31T19:05:37.28881Z","iopub.status.idle":"2021-07-31T19:17:34.275909Z","shell.execute_reply":"2021-07-31T19:17:34.277406Z","shell.execute_reply.started":"2021-07-31T18:02:18.453437Z"},"papermill":{"duration":717.241299,"end_time":"2021-07-31T19:17:34.277694","exception":false,"start_time":"2021-07-31T19:05:37.036395","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params3 = {'bagging_fraction': 0.63, 'boosting': 'gbdt', 'feature_fraction': 0.8, 'lambda_l1': 2.5188543779297663,\n          'lambda_l2': 1.1247539949521028, 'learning_rate': 0.022110756156901078, 'max_bin': 217, 'max_depth': 7,\n          'metric': 'mae', 'min_data_in_bin': 165, 'min_data_in_leaf': 169, 'min_gain_to_split': 2.47, 'num_leaves': 128,\n          'objective': 'mae', 'subsample': 0.6468413409317881}\nmodels3_, oof3_ = fit_lgbm_(\n    train, feature_cols, \"target3\", params3\n)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T19:17:34.816987Z","iopub.status.busy":"2021-07-31T19:17:34.816137Z","iopub.status.idle":"2021-07-31T19:31:54.943562Z","shell.execute_reply":"2021-07-31T19:31:54.944574Z","shell.execute_reply.started":"2021-07-31T18:14:45.685406Z"},"papermill":{"duration":860.40206,"end_time":"2021-07-31T19:31:54.944854","exception":false,"start_time":"2021-07-31T19:17:34.542794","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params4 = {'bagging_fraction': 0.53, 'boosting': 'gbdt', 'feature_fraction': 0.5, 'lambda_l1': 2.3658020794556784,\n           'lambda_l2': 1.36597241600689, 'learning_rate': 0.11612709577750321, 'max_bin': 150, 'max_depth': 13,\n           'metric': 'mae', 'min_data_in_bin': 89, 'min_data_in_leaf': 223, 'min_gain_to_split': 0.63, 'num_leaves': 86, \n           'objective': 'fair', 'subsample': 0.8314126324024551}\nmodels4_, oof4_ = fit_lgbm_(\n    train, feature_cols, \"target4\", params4\n)","metadata":{"execution":{"iopub.execute_input":"2021-07-31T19:31:55.504938Z","iopub.status.busy":"2021-07-31T19:31:55.504299Z","iopub.status.idle":"2021-07-31T19:35:01.881707Z","shell.execute_reply":"2021-07-31T19:35:01.882377Z","shell.execute_reply.started":"2021-07-31T18:31:15.785869Z"},"papermill":{"duration":186.65724,"end_time":"2021-07-31T19:35:01.882593","exception":false,"start_time":"2021-07-31T19:31:55.225353","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_map = dict()\n\nfor player in tqdm(train.playerId.unique()):\n    value = train.daysAfterGame[train.playerId == player].values[-1]\n    player_map[player] = value","metadata":{"execution":{"iopub.execute_input":"2021-07-31T19:35:02.452576Z","iopub.status.busy":"2021-07-31T19:35:02.451938Z","iopub.status.idle":"2021-07-31T19:35:04.116214Z","shell.execute_reply":"2021-07-31T19:35:04.115556Z","shell.execute_reply.started":"2021-07-31T18:34:12.147827Z"},"papermill":{"duration":1.950013,"end_time":"2021-07-31T19:35:04.116353","exception":false,"start_time":"2021-07-31T19:35:02.16634","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances', \"positionCode\"\n]\n\nnull = np.nan\ntrue = True\nfalse = False","metadata":{"execution":{"iopub.execute_input":"2021-07-31T19:35:04.701843Z","iopub.status.busy":"2021-07-31T19:35:04.701088Z","iopub.status.idle":"2021-07-31T19:35:04.703611Z","shell.execute_reply":"2021-07-31T19:35:04.704121Z","shell.execute_reply.started":"2021-07-31T18:34:13.777791Z"},"papermill":{"duration":0.30091,"end_time":"2021-07-31T19:35:04.704291","exception":false,"start_time":"2021-07-31T19:35:04.403381","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rank_dict = {\n    4: 3,\n    5: 4,\n    6: 5,\n    7: 6,\n    8: 7,\n    9: 8,\n    10: 9\n    \n}","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:35:25.436127Z","iopub.status.busy":"2021-07-31T18:35:25.435737Z","iopub.status.idle":"2021-07-31T18:35:25.441136Z","shell.execute_reply":"2021-07-31T18:35:25.439891Z","shell.execute_reply.started":"2021-07-31T18:35:25.436092Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import copy\n\n# mlb.make_env.__called__ = False\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in tqdm(iter_test): # make predictions here\n    \n    sub = copy.deepcopy(sample_prediction_df.reset_index())\n    sample_prediction_df = copy.deepcopy(sample_prediction_df.reset_index(drop=True))\n    \n    # LGBM summit\n    # creat dataset\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    # Dealing with missing values\n    if test_df['rosters'].iloc[0] == test_df['rosters'].iloc[0]:\n        test_rosters = pd.DataFrame(eval(test_df['rosters'].iloc[0]))\n    else:\n        test_rosters = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in rosters.columns:\n            if col == 'playerId': continue\n            test_rosters[col] = np.nan\n            \n    if test_df['playerBoxScores'].iloc[0] == test_df['playerBoxScores'].iloc[0]:\n        test_scores = pd.DataFrame(eval(test_df['playerBoxScores'].iloc[0]))\n        test_scores[\"positionCode\"] = test_scores[\"positionCode\"].astype(np.int32)\n    else:\n        test_scores = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in scores.columns:\n            if col == 'playerId': continue\n            test_scores[col] = np.nan\n    test_scores = test_scores.groupby('playerId').sum().reset_index()\n    test = sample_prediction_df[['playerId']].copy()\n    test = test.merge(players[players_cols], on='playerId', how='left')\n    test = test.merge(test_rosters[rosters_cols], on='playerId', how='left')\n    test = test.merge(test_scores[scores_cols], on='playerId', how='left')\n    test = test.merge(player_target_stats, how='left', left_on=[\"playerId\"],right_on=[\"playerId\"])\n    \n\n    test['label_playerId'] = test['playerId'].map(player2num)\n    test['label_primaryPositionName'] = test['primaryPositionName'].map(position2num)\n    test['label_teamId'] = test['teamId'].map(teamid2num)\n    test['label_status'] = test['status'].map(status2num)\n    \n    # print(test.shape)    \n    all_cols = test.columns.values.tolist()\n    test['year'] = 2021\n    test = pd.merge(test,\n                       seasons_df,\n                       how = 'left',\n                       left_on = 'year',\n                       right_on = 'seasonId')\n    test['days_to_season_start'] = (test['seasonStartDate'] - pd.to_datetime(test_df.index.values[0], format = '%Y%m%d')).dt.days\n    all_cols.append('days_to_season_start')\n    test = test[all_cols]   \n    # make daysAfterGame\n    test[\"isGame\"] = test[\"positionCode\"].notna().astype(np.int32)\n    \n    for player, value in player_map.items():\n        if player in test.playerId.unique():\n            is_game = test.isGame[test.playerId == player].values[-1]\n            if is_game == 1:\n                player_map[player] = 0\n            elif is_game == 0:\n                player_map[player] = value + 1\n        else:\n            player_map[player] = value + 1\n    \n    test_ = pd.DataFrame()\n    \n    test_[\"playerId\"] = list(player_map.keys())\n    test_[\"daysAfterGame\"] = list(player_map.values())\n    \n    test = test.merge(test_, how=\"left\", on=\"playerId\")\n    # make daysAfterGame\n    test['year'] = 2021\n    test = test.merge(stats.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])\n    test = test.merge(stats_team.reset_index(), how=\"left\", on=[\"teamId\", \"year\"])\n    test = test.merge(stats_code.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])\n    test[\"rankMonth\"] = rank_dict[pd.to_datetime(test_df.index.values[0], format = '%Y%m%d').month]\n    test = test.merge(stats1.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])\n    test = test.merge(stats2.reset_index(), how=\"left\", on=[\"playerId\", \"year\"])\n        \n    test_X = test[feature_cols]\n\n    pred1_lgb = []\n    for model in models1_:\n        pred1_lgb.append(model.predict(test_X))\n    pred1_lgb = np.mean(pred1_lgb, axis=0)\n    \n    pred2_lgb = []\n    for model in models2_:\n        pred2_lgb.append(model.predict(test_X))\n    pred2_lgb = np.mean(pred2_lgb, axis=0)\n    \n    \n    pred3_lgb = []\n    for model in models3_:\n        pred3_lgb.append(model.predict(test_X))\n    pred3_lgb = np.mean(pred3_lgb, axis=0)\n    \n    pred4_lgb = []\n    for model in models4_:\n        pred4_lgb.append(model.predict(test_X))\n    pred4_lgb = np.mean(pred4_lgb, axis=0)\n\n    \n\n    sample_prediction_df['target1'] = np.clip(pred1_lgb, 0, 100)\n    sample_prediction_df['target2'] = np.clip(pred2_lgb, 0, 100)\n    sample_prediction_df['target3'] = np.clip(pred3_lgb, 0, 100)\n    sample_prediction_df['target4'] = np.clip(pred4_lgb, 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)\n    del sample_prediction_df['playerId']\n\n    env.predict(sample_prediction_df)\n","metadata":{"execution":{"iopub.execute_input":"2021-07-31T18:35:29.528664Z","iopub.status.busy":"2021-07-31T18:35:29.52825Z","iopub.status.idle":"2021-07-31T18:35:38.302912Z","shell.execute_reply":"2021-07-31T18:35:38.301982Z","shell.execute_reply.started":"2021-07-31T18:35:29.528626Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]},"execution_count":null,"outputs":[]}]}