{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div>\n    <h1 align=\"center\">MLB Player Digital Engagement Forecasting</h1>\n    <h1 align=\"center\">LightGBM + CatBoost + ANN 2505f2</h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">\n    <h1 align=\"center\">If you find this work useful, please don't forget upvoting :)</h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"#### Thanks to: @lhagiimn   https://www.kaggle.com/lhagiimn/lightgbm-catboost-ann-2505f2\n\n#### https://www.kaggle.com/columbia2131/mlb-lightgbm-starter-dataset-code-en-ja\n\n#### https://www.kaggle.com/mlconsult/1-3816-lb-lbgm-descriptive-stats-param-tune\n\n#### https://www.kaggle.com/batprem/lightgbm-ann-weight-with-love\n\n#### https://www.kaggle.com/mlconsult/1-3816-lb-lbgm-descriptive-stats-param-tune\n\n#### https://www.kaggle.com/ulrich07/mlb-ann-with-lags-tf-keras\n","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"markdown","source":"## About Dataset","metadata":{"_uuid":"dedc4e57-daea-4f1f-9281-257352546646","_cell_guid":"7cda4356-b015-4992-877b-e40eb158b01e","trusted":true}},{"cell_type":"markdown","source":"Train.csv is stored as a csv file with each column as follows.  \n\ntrain.csvを以下のようにして各カラムをcsvファイルとして保管しています。","metadata":{"_uuid":"02d49fe6-62ea-4eed-9595-cddd3dee465e","_cell_guid":"8c3c5497-d485-47c0-99af-b2f8169cdca4","trusted":true}},{"cell_type":"code","source":"!cp ../input/fork-of-1-35-lightgbm-ann-2505f2-c4e96a/* .","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:24:17.727036Z","iopub.execute_input":"2021-07-31T03:24:17.727477Z","iopub.status.idle":"2021-07-31T03:24:19.716454Z","shell.execute_reply.started":"2021-07-31T03:24:17.727441Z","shell.execute_reply":"2021-07-31T03:24:19.715339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:24:21.523426Z","iopub.execute_input":"2021-07-31T03:24:21.523839Z","iopub.status.idle":"2021-07-31T03:24:23.340031Z","shell.execute_reply.started":"2021-07-31T03:24:21.523805Z","shell.execute_reply":"2021-07-31T03:24:23.338811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\"\"\"\n!pip install pandarallel \n\nimport gc\n\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\nfrom pandarallel import pandarallel\npandarallel.initialize()\n\nBASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\ntrain = pd.read_csv(BASE_DIR / 'train.csv')\n\nnull = np.nan\ntrue = True\nfalse = False\n\nfor col in train.columns:\n\n    if col == 'date': continue\n\n    _index = train[col].notnull()\n    train.loc[_index, col] = train.loc[_index, col].parallel_apply(lambda x: eval(x))\n\n    outputs = []\n    for index, date, record in train.loc[_index, ['date', col]].itertuples():\n        _df = pd.DataFrame(record)\n        _df['index'] = index\n        _df['date'] = date\n        outputs.append(_df)\n\n    outputs = pd.concat(outputs).reset_index(drop=True)\n\n    outputs.to_csv(f'{col}_train.csv', index=False)\n    outputs.to_pickle(f'{col}_train.pkl')\n\n    del outputs\n    del train[col]\n    gc.collect()\n\"\"\"","metadata":{"_uuid":"62aca0d3-6af3-4760-9db0-0a397fdc5191","_cell_guid":"733fb600-c8da-4658-aa35-320c6817a8c5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:24.237582Z","iopub.execute_input":"2021-07-31T03:24:24.237942Z","iopub.status.idle":"2021-07-31T03:24:26.402687Z","shell.execute_reply.started":"2021-07-31T03:24:24.237909Z","shell.execute_reply":"2021-07-31T03:24:26.401825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{"_uuid":"ea4d85e1-21d7-4e7a-a8e7-36e5f53f4612","_cell_guid":"8d708767-e50d-4684-82a1-790feb5c0c1f","execution":{"iopub.status.busy":"2021-06-16T09:14:33.869464Z","iopub.execute_input":"2021-06-16T09:14:33.869905Z","iopub.status.idle":"2021-06-16T09:14:33.874766Z","shell.execute_reply.started":"2021-06-16T09:14:33.869879Z","shell.execute_reply":"2021-06-16T09:14:33.873097Z"},"trusted":true}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.metrics import mean_absolute_error\nfrom datetime import timedelta\nfrom functools import reduce\nfrom tqdm import tqdm\nimport lightgbm as lgbm\nimport mlb\nimport os","metadata":{"_uuid":"2139878b-da24-41e3-bb59-76b60a1e16ef","_cell_guid":"3181642b-6cb0-424d-b5be-a0e6a5cb0457","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:26.403967Z","iopub.execute_input":"2021-07-31T03:24:26.404405Z","iopub.status.idle":"2021-07-31T03:24:27.476363Z","shell.execute_reply.started":"2021-07-31T03:24:26.404371Z","shell.execute_reply":"2021-07-31T03:24:27.475301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\nTRAIN_DIR = Path('../input/mlb-pdef-train-dataset')","metadata":{"_uuid":"3540a2ce-95e1-416f-9892-bcfa92cf6047","_cell_guid":"5f9dc680-5158-4bf2-856f-d43b6aa620de","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:27.549108Z","iopub.execute_input":"2021-07-31T03:24:27.549507Z","iopub.status.idle":"2021-07-31T03:24:28.755458Z","shell.execute_reply.started":"2021-07-31T03:24:27.54947Z","shell.execute_reply":"2021-07-31T03:24:28.754484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv(BASE_DIR / 'players.csv')\n\nrosters = pd.read_pickle(TRAIN_DIR / 'rosters_train.pkl')\ntargets = pd.read_pickle(TRAIN_DIR / 'nextDayPlayerEngagement_train.pkl')\nscores = pd.read_pickle(TRAIN_DIR / 'playerBoxScores_train.pkl')\nscores = scores.groupby(['playerId', 'date']).sum().reset_index()","metadata":{"_uuid":"38e056cf-cf8f-45f0-9ad4-cda6d11917a2","_cell_guid":"e67cda79-ae71-474a-861b-2b437b94a0a8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:28.756972Z","iopub.execute_input":"2021-07-31T03:24:28.757278Z","iopub.status.idle":"2021-07-31T03:24:33.309918Z","shell.execute_reply.started":"2021-07-31T03:24:28.757248Z","shell.execute_reply":"2021-07-31T03:24:33.308637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:24:33.314081Z","iopub.execute_input":"2021-07-31T03:24:33.314445Z","iopub.status.idle":"2021-07-31T03:24:34.399806Z","shell.execute_reply.started":"2021-07-31T03:24:33.314413Z","shell.execute_reply":"2021-07-31T03:24:34.399026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets_cols = ['playerId', 'target1', 'target2', 'target3', 'target4', 'date']\nplayers_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status', 'date']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances', 'date']\n\nfeature_cols = ['label_playerId', 'label_primaryPositionName', 'label_teamId',\n       'label_status', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances','target1_mean',\n 'target1_median',\n 'target1_std',\n 'target1_min',\n 'target1_max',\n 'target1_prob',\n 'target2_mean',\n 'target2_median',\n 'target2_std',\n 'target2_min',\n 'target2_max',\n 'target2_prob',\n 'target3_mean',\n 'target3_median',\n 'target3_std',\n 'target3_min',\n 'target3_max',\n 'target3_prob',\n 'target4_mean',\n 'target4_median',\n 'target4_std',\n 'target4_min',\n 'target4_max',\n 'target4_prob']\nfeature_cols2 = ['label_playerId', 'label_primaryPositionName', 'label_teamId',\n       'label_status', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances','target1_mean',\n 'target1_median',\n 'target1_std',\n 'target1_min',\n 'target1_max',\n 'target1_prob',\n 'target2_mean',\n 'target2_median',\n 'target2_std',\n 'target2_min',\n 'target2_max',\n 'target2_prob',\n 'target3_mean',\n 'target3_median',\n 'target3_std',\n 'target3_min',\n 'target3_max',\n 'target3_prob',\n 'target4_mean',\n 'target4_median',\n 'target4_std',\n 'target4_min',\n 'target4_max',\n 'target4_prob',\n    'target1']","metadata":{"_uuid":"046564c8-2d25-4540-9a94-9e629e263a22","_cell_guid":"39e17ba0-96e5-438c-94bd-6c184905c35f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:34.401252Z","iopub.execute_input":"2021-07-31T03:24:34.401649Z","iopub.status.idle":"2021-07-31T03:24:35.509666Z","shell.execute_reply.started":"2021-07-31T03:24:34.401619Z","shell.execute_reply":"2021-07-31T03:24:35.508691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_target_stats = pd.read_csv(\"../input/player-target-stats/player_target_stats.csv\")\ndata_names=player_target_stats.columns.values.tolist()\ndata_names","metadata":{"_uuid":"09b65c0a-4ac9-45cb-86ee-d81e871960a7","_cell_guid":"5ee50f59-b249-4698-ae86-e35289c6df01","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:35.513305Z","iopub.execute_input":"2021-07-31T03:24:35.513625Z","iopub.status.idle":"2021-07-31T03:24:36.62056Z","shell.execute_reply.started":"2021-07-31T03:24:35.513595Z","shell.execute_reply":"2021-07-31T03:24:36.619561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creat dataset\ntrain = targets[targets_cols].merge(players[players_cols], on=['playerId'], how='left')\ntrain = train.merge(rosters[rosters_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(scores[scores_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(player_target_stats, how='inner', left_on=[\"playerId\"],right_on=[\"playerId\"])\n\n\n# label encoding\nplayer2num = {c: i for i, c in enumerate(train['playerId'].unique())}\nposition2num = {c: i for i, c in enumerate(train['primaryPositionName'].unique())}\nteamid2num = {c: i for i, c in enumerate(train['teamId'].unique())}\nstatus2num = {c: i for i, c in enumerate(train['status'].unique())}\ntrain['label_playerId'] = train['playerId'].map(player2num)\ntrain['label_primaryPositionName'] = train['primaryPositionName'].map(position2num)\ntrain['label_teamId'] = train['teamId'].map(teamid2num)\ntrain['label_status'] = train['status'].map(status2num)","metadata":{"_uuid":"6cb3c4b2-a69b-4ce5-9d70-6633354c11ea","_cell_guid":"69624101-e45d-442a-bbb4-fb5514840e21","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:36.621835Z","iopub.execute_input":"2021-07-31T03:24:36.622124Z","iopub.status.idle":"2021-07-31T03:24:46.029817Z","shell.execute_reply.started":"2021-07-31T03:24:36.622096Z","shell.execute_reply":"2021-07-31T03:24:46.028613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[feature_cols]\ntrain_y = train[['target1', 'target2', 'target3', 'target4']]\n\n_index = (train['date'] > 20210401)\nx_train1 = train_X.loc[_index].reset_index(drop=True)\ny_train1 = train_y.loc[_index].reset_index(drop=True)\nx_valid1 = train_X.loc[~_index].reset_index(drop=True)\ny_valid1 = train_y.loc[~_index].reset_index(drop=True)","metadata":{"_uuid":"ddc79e19-ca6c-40d2-bb71-cf5df58549ce","_cell_guid":"ffc2b454-9e48-44f8-b4fa-d2d87ea5e89d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:24:46.031312Z","iopub.execute_input":"2021-07-31T03:24:46.031743Z","iopub.status.idle":"2021-07-31T03:24:51.727037Z","shell.execute_reply.started":"2021-07-31T03:24:46.0317Z","shell.execute_reply":"2021-07-31T03:24:51.726298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"len(y_valid1)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:24:51.728685Z","iopub.execute_input":"2021-07-31T03:24:51.729117Z","iopub.status.idle":"2021-07-31T03:24:52.856635Z","shell.execute_reply.started":"2021-07-31T03:24:51.729084Z","shell.execute_reply":"2021-07-31T03:24:52.855779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[feature_cols2]\ntrain_y = train[['target1', 'target2', 'target3', 'target4']]\n\n_index = (train['date'] > 20210401)\nx_train2 = train_X.loc[_index].reset_index(drop=True)\ny_train2 = train_y.loc[_index].reset_index(drop=True)\nx_valid2 = train_X.loc[~_index].reset_index(drop=True)\ny_valid2 = train_y.loc[~_index].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:24:52.8577Z","iopub.execute_input":"2021-07-31T03:24:52.8581Z","iopub.status.idle":"2021-07-31T03:24:56.387939Z","shell.execute_reply.started":"2021-07-31T03:24:52.858068Z","shell.execute_reply":"2021-07-31T03:24:56.38704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install wandb","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:39:38.767655Z","iopub.execute_input":"2021-07-31T04:39:38.768118Z","iopub.status.idle":"2021-07-31T04:39:46.338247Z","shell.execute_reply.started":"2021-07-31T04:39:38.768023Z","shell.execute_reply":"2021-07-31T04:39:46.337069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\nfrom wandb.keras import WandbCallback\n\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:39:51.533638Z","iopub.execute_input":"2021-07-31T04:39:51.534017Z","iopub.status.idle":"2021-07-31T04:40:11.352888Z","shell.execute_reply.started":"2021-07-31T04:39:51.533974Z","shell.execute_reply":"2021-07-31T04:40:11.351979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_lgbm(x_train, y_train, x_valid, y_valid, params: dict=None, verbose=100):\n    oof_pred = np.zeros(len(y_valid), dtype=np.float32)\n    model = lgbm.LGBMRegressor(**params)\n    model.fit(x_train, y_train, \n        eval_set=[(x_valid, y_valid)],  \n        early_stopping_rounds=verbose, \n        verbose=verbose)\n    oof_pred = model.predict(x_valid)\n    score = mean_absolute_error(oof_pred, y_valid)\n    print('mae:', score)\n#     return oof_pred, model, score\n\n\n# training lightgbm\n\nparams1 = {'objective':'mae',\n           'reg_alpha': 0.14947461820098767, \n           'reg_lambda': 0.10185644384043743, \n           'n_estimators': 3633, \n           'learning_rate': 0.08046301304430488, \n           'num_leaves': 674, \n           'feature_fraction': 0.9101240539122566, \n           'bagging_fraction': 0.9884451442950513, \n           'bagging_freq': 8, \n           'min_child_samples': 51}\n\nparams2 = {\n 'objective':'mae',\n 'reg_alpha': 0.1,\n 'reg_lambda': 0.1, \n 'n_estimators': 80,\n 'learning_rate': 0.1,\n 'random_state': 42,\n \"num_leaves\": 22\n}\n\nparams4 = {'objective':'mae',\n           'reg_alpha': 0.016468100279441976, \n           'reg_lambda': 0.09128335764019105, \n           'n_estimators': 9868, \n           'learning_rate': 0.10528150510326864, \n           'num_leaves': 157, \n           'feature_fraction': 0.5419185713426886, \n           'bagging_fraction': 0.2637405128936662, \n           'bagging_freq': 19, \n           'min_child_samples': 71}\n\n\nparams = {\n 'objective':'mae',\n 'reg_alpha': 0.1,\n 'reg_lambda': 0.1, \n 'n_estimators': 10000,\n 'learning_rate': 0.1,\n 'random_state': 42,\n \"num_leaves\": 100\n}\n\n\n# oof1, model1, score1 = fit_lgbm(\n#     x_train1, y_train1['target1'],\n#     x_valid1, y_valid1['target1'],\n#     params1\n#  )\n\n# oof2, model2, score2 = fit_lgbm(\n#     x_train2, y_train2['target2'],\n#     x_valid2, y_valid2['target2'],\n#     params2\n# )\n\n# oof3, model3, score3 = fit_lgbm(\n#     x_train2, y_train2['target3'],\n#     x_valid2, y_valid2['target3'],\n#    params\n# )\n\n# oof4, model4, score4 = fit_lgbm(\n#     x_train2, y_train2['target4'],\n#     x_valid2, y_valid2['target4'],\n#     params4\n# )\n\n# score = (score1+score2+score3+score4) / 4\n# print(f'score: {score}')\n\n\n","metadata":{"_uuid":"c616b4b9-961c-4fcc-9643-391e8225a43c","_cell_guid":"50c7c4dd-9076-4e0a-98bc-bb2e56f0ecb4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T04:40:13.924289Z","iopub.execute_input":"2021-07-31T04:40:13.924854Z","iopub.status.idle":"2021-07-31T04:40:13.934898Z","shell.execute_reply.started":"2021-07-31T04:40:13.924813Z","shell.execute_reply":"2021-07-31T04:40:13.933825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sweep_config1 = {\n    # 'method': 'grid',\n    'method':'random',\n    'metric': {\n      'name': 'mae',\n      'goal': 'minimize'\n  },\n    'early_terminate':{\n        'type': 'hyperband',\n        'min_iter': 5\n      }\n}","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:40:17.452691Z","iopub.execute_input":"2021-07-31T04:40:17.453279Z","iopub.status.idle":"2021-07-31T04:40:17.457815Z","shell.execute_reply.started":"2021-07-31T04:40:17.453241Z","shell.execute_reply":"2021-07-31T04:40:17.456848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config_defaults1 = {\n    'model_name':'model1',\n        'objective':'mae',\n           'reg_alpha': 0.14947461820098767, \n           'reg_lambda': 0.10185644384043743, \n           'n_estimators': 3633, \n           'learning_rate': 0.08046301304430488, \n           'num_leaves': 674, \n           'feature_fraction': 0.9101240539122566, \n           'bagging_fraction': 0.9884451442950513, \n           'bagging_freq': 8, \n           'min_child_samples': 51,\n        }\n\nparameters_1_dict = {\n    'learning_rate': {\n        'distribution':\"categorical\",\n        'values': [0.0001,0.00015,0.0005,0.001,0.005,0.01,0.05]\n        },\n    'reg_lambda': {\n        'distribution':\"categorical\",\n        'values': [0.0001,0.00015,0.0005,0.001,0.005,0.01,0.05]\n        },\n    'reg_lambda':{\n        'distribution':\"categorical\",\n        'values': [0.0001,0.00015,0.0005,0.001,0.005,0.01,0.05]\n    },\n    'num_leaves': {\n        'distribution':\"uniform\",\n        'max': 1000,\n        'min':10\n        },\n    'feature_fraction': {\n        'distribution':\"uniform\",\n        'max': 1,\n        'min':0.1\n        },\n    'bagging_fraction': {\n        'distribution':\"uniform\",\n        'max': 1,\n        'min':0.1\n        },\n    \n    }\n","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:40:18.52897Z","iopub.execute_input":"2021-07-31T04:40:18.529312Z","iopub.status.idle":"2021-07-31T04:40:18.536797Z","shell.execute_reply.started":"2021-07-31T04:40:18.529278Z","shell.execute_reply":"2021-07-31T04:40:18.535563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sweep_config1['parameters'] = parameters_1_dict","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:40:19.615428Z","iopub.execute_input":"2021-07-31T04:40:19.615786Z","iopub.status.idle":"2021-07-31T04:40:19.620551Z","shell.execute_reply.started":"2021-07-31T04:40:19.615752Z","shell.execute_reply":"2021-07-31T04:40:19.619045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.init(config=config_defaults1)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:40:20.852921Z","iopub.execute_input":"2021-07-31T04:40:20.853552Z","iopub.status.idle":"2021-07-31T04:40:24.715986Z","shell.execute_reply.started":"2021-07-31T04:40:20.853501Z","shell.execute_reply":"2021-07-31T04:40:24.714716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture \npip install wandb --upgrade","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:42:40.644852Z","iopub.execute_input":"2021-07-31T04:42:40.645177Z","iopub.status.idle":"2021-07-31T04:42:51.299915Z","shell.execute_reply.started":"2021-07-31T04:42:40.645149Z","shell.execute_reply":"2021-07-31T04:42:51.298795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit_lgbm(\n#     x_train1, y_train1['target1'],\n#     x_valid1, y_valid1['target1'],\n#     params1\n#  )","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:18:23.347903Z","iopub.execute_input":"2021-07-31T03:18:23.348387Z","iopub.status.idle":"2021-07-31T03:18:23.353349Z","shell.execute_reply.started":"2021-07-31T03:18:23.348344Z","shell.execute_reply":"2021-07-31T03:18:23.352107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sweep_train():\n    config_defaults1 = {\n    'model_name':'model1',\n        'objective':'mae',\n           'reg_alpha': 0.14947461820098767, \n           'reg_lambda': 0.10185644384043743, \n           'n_estimators': 3633, \n           'learning_rate': 0.08046301304430488, \n           'num_leaves': 674, \n           'feature_fraction': 0.9101240539122566, \n           'bagging_fraction': 0.9884451442950513, \n           'bagging_freq': 8, \n           'min_child_samples': 51,\n        }\n    print(\"hi\")\n    wandb.init(config=config_defaults1)\n    \n    params1 = {'objective':wandb.config.objective,\n           'reg_alpha': wandb.config.reg_alpha, \n           'reg_lambda': wandb.config.reg_lambda, \n           'n_estimators': wandb.config.n_estimators, \n           'learning_rate': wandb.config.learning_rate, \n           'num_leaves': wandb.config.num_leaves, \n           'feature_fraction': wandb.config.feature_fraction, \n           'bagging_fraction': wandb.config.bagging_fraction, \n           'bagging_freq': wandb.config.bagging_freq, \n           'min_child_samples': wandb.config.min_child_samples}\n    \n    print(\"hi\")\n    oof_pred = np.zeros(len(y_train1['target1']), dtype=np.float32)\n    model = lgbm.LGBMRegressor(**params)\n    model.fit(x_train1, y_train1['target1'], \n        eval_set= [(x_valid1, y_valid1['target1'])],  \n        early_stopping_rounds=100, \n        verbose=100)\n    oof_pred = model.predict(x_valid1)\n    wandb.config.score = mean_absolute_error(oof_pred, y_valid1['target1'])\n    print(str(score))\n    wandb.log({\n        'objective':wandb.config.objective,\n#            'reg_alpha': wandb.config.reg_alpha, \n#            'reg_lambda': wandb.config.reg_lambda, \n#            'n_estimators': wandb.config.n_estimators, \n#            'learning_rate': wandb.config.learning_rate, \n#            'num_leaves': wandb.config.num_leaves, \n#            'feature_fraction': wandb.config.feature_fraction, \n#            'bagging_fraction': wandb.config.bagging_fraction, \n#            'bagging_freq': wandb.config.bagging_freq, \n#            'min_child_samples': wandb.config.min_child_samples,\n            'mae':  wandb.config.score\n    })\n    \n#     fit_lgbm(\n#             x_train1, y_train1['target1'],\n#             x_valid1, y_valid1['target1'],\n#             params1\n#          )","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:40:32.112596Z","iopub.execute_input":"2021-07-31T04:40:32.112936Z","iopub.status.idle":"2021-07-31T04:40:32.694968Z","shell.execute_reply.started":"2021-07-31T04:40:32.112904Z","shell.execute_reply":"2021-07-31T04:40:32.693375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit_lgbm(\n#     x_train1, y_train1['target1'],\n#     x_valid1, y_valid1['target1'],\n#     params1\n#  )","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:40:33.273431Z","iopub.execute_input":"2021-07-31T04:40:33.273801Z","iopub.status.idle":"2021-07-31T04:40:33.782322Z","shell.execute_reply.started":"2021-07-31T04:40:33.273768Z","shell.execute_reply":"2021-07-31T04:40:33.781549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\n","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:43:05.899174Z","iopub.execute_input":"2021-07-31T04:43:05.899579Z","iopub.status.idle":"2021-07-31T04:43:06.456969Z","shell.execute_reply.started":"2021-07-31T04:43:05.899538Z","shell.execute_reply":"2021-07-31T04:43:06.456227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sweep_id = wandb.sweep(sweep_config1)# project=\"MLB_hyperparam\"","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:43:07.491994Z","iopub.execute_input":"2021-07-31T04:43:07.492572Z","iopub.status.idle":"2021-07-31T04:43:08.150775Z","shell.execute_reply.started":"2021-07-31T04:43:07.492518Z","shell.execute_reply":"2021-07-31T04:43:08.149114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.agent(sweep_id, function=sweep_train, count=30)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:43:09.04687Z","iopub.execute_input":"2021-07-31T04:43:09.047201Z","iopub.status.idle":"2021-07-31T04:43:09.579829Z","shell.execute_reply.started":"2021-07-31T04:43:09.04717Z","shell.execute_reply":"2021-07-31T04:43:09.578322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nfrom catboost import CatBoostRegressor\n\ndef fit_lgbm(x_train, y_train, x_valid, y_valid, target, params: dict=None, verbose=100):\n    oof_pred_lgb = np.zeros(len(y_valid), dtype=np.float32)\n    oof_pred_cat = np.zeros(len(y_valid), dtype=np.float32)\n    \n    if os.path.isfile(f'../input/mlb-lgbm-and-catboost-models/model_lgb_{target}.pkl'):\n        with open(f'../input/mlb-lgbm-and-catboost-models/model_lgb_{target}.pkl', 'rb') as fin:\n            model = pickle.load(fin)\n    else:\n    \n        model = lgbm.LGBMRegressor(**params)\n        model.fit(x_train, y_train, \n            eval_set=[(x_valid, y_valid)],  \n            early_stopping_rounds=verbose, \n            verbose=verbose)\n\n        with open(f'model_lgb_{target}.pkl', 'wb') as handle:\n            pickle.dump(model, handle, protocol=pickle.HIGHEST_PROTOCOL)\n    \n    oof_pred_lgb = model.predict(x_valid)\n    score_lgb = mean_absolute_error(oof_pred_lgb, y_valid)\n    print('mae:', score_lgb)\n    \n    if os.path.isfile(f'../input/mlb-lgbm-and-catboost-models/model_cb_{target}.pkl'):\n        with open(f'../input/mlb-lgbm-and-catboost-models/model_cb_{target}.pkl', 'rb') as fin:\n            model_cb = pickle.load(fin)\n    else:\n    \n        model_cb = CatBoostRegressor(\n                    n_estimators=2000,\n                    learning_rate=0.05,\n                    loss_function='MAE',\n                    eval_metric='MAE',\n                    max_bin=50,\n                    subsample=0.9,\n                    colsample_bylevel=0.5,\n                    verbose=100)\n\n        model_cb.fit(x_train, y_train, use_best_model=True,\n                         eval_set=(x_valid, y_valid),\n                         early_stopping_rounds=25)\n\n        with open(f'model_cb_{target}.pkl', 'wb') as handle:\n            pickle.dump(model_cb, handle, protocol=pickle.HIGHEST_PROTOCOL)\n    \n    oof_pred_cat = model_cb.predict(x_valid)\n    score_cat = mean_absolute_error(oof_pred_cat, y_valid)\n    print('mae:', score_cat)\n    \n    return oof_pred_lgb, model, oof_pred_cat, model_cb, score_lgb, score_cat\n\n\n# training lightgbm\nparams = {\n'boosting_type': 'gbdt',\n'objective':'mae',\n'subsample': 0.5,\n'subsample_freq': 1,\n'learning_rate': 0.03,\n'num_leaves': 2**11-1,\n'min_data_in_leaf': 2**12-1,\n'feature_fraction': 0.5,\n'max_bin': 100,\n'n_estimators': 2500,\n'boost_from_average': False,\n\"random_seed\":42,\n}\n\noof_pred_lgb2, model_lgb2, oof_pred_cat2, model_cb2, score_lgb2, score_cat2 = fit_lgbm(\n    x_train1, y_train1['target2'],\n    x_valid1, y_valid1['target2'],\n    2, params\n)\n\noof_pred_lgb1, model_lgb1, oof_pred_cat1, model_cb1, score_lgb1, score_cat1 = fit_lgbm(\n    x_train1, y_train1['target1'],\n    x_valid1, y_valid1['target1'],\n    1, params\n)\n\noof_pred_lgb3, model_lgb3, oof_pred_cat3, model_cb3, score_lgb3, score_cat3 = fit_lgbm(\n    x_train1, y_train1['target3'],\n    x_valid1, y_valid1['target3'],\n    3, params\n)\noof_pred_lgb4, model_lgb4, oof_pred_cat4, model_cb4, score_lgb4, score_cat4= fit_lgbm(\n    x_train1, y_train1['target4'],\n    x_valid1, y_valid1['target4'],\n    4, params\n)\n\nscore = (score_lgb1+score_lgb2+score_lgb3+score_lgb4) / 4\nprint(f'LightGBM score: {score}')\n\nscore = (score_cat1+score_cat2+score_cat3+score_cat4) / 4\nprint(f'Catboost score: {score}')","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:26:24.878937Z","iopub.execute_input":"2021-07-31T03:26:24.879348Z","iopub.status.idle":"2021-07-31T03:26:32.834575Z","shell.execute_reply.started":"2021-07-31T03:26:24.879311Z","shell.execute_reply":"2021-07-31T03:26:32.832443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{"_uuid":"372a2718-6f2d-4128-97cb-fefd653e1f40","_cell_guid":"25cf48a6-84c3-4d5d-a312-1a47c050722e","trusted":true}},{"cell_type":"code","source":"players_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances']\n\nnull = np.nan\ntrue = True\nfalse = False","metadata":{"_uuid":"7828a113-8dc7-4691-b2a6-b232ca0525dc","_cell_guid":"08cf9a6e-1f2c-4acd-b714-703d641d996a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:15:25.493386Z","iopub.status.idle":"2021-07-31T03:15:25.494094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datetime import timedelta\nfrom tqdm import tqdm\nimport gc\nfrom functools import reduce\nfrom sklearn.model_selection import StratifiedKFold\n\nROOT_DIR = \"../input/mlb-player-digital-engagement-forecasting\"\n\n#=======================#\ndef flatten(df, col):\n    du = (df.pivot(index=\"playerId\", columns=\"EvalDate\", \n               values=col).add_prefix(f\"{col}_\").\n      rename_axis(None, axis=1).reset_index())\n    return du\n#============================#\ndef reducer(left, right):\n    return left.merge(right, on=\"playerId\")\n#========================\n\nTGTCOLS = [\"target1\",\"target2\",\"target3\",\"target4\"]\ndef train_lag(df, lag=1):\n    dp = df[[\"playerId\",\"EvalDate\"]+TGTCOLS].copy()\n    dp[\"EvalDate\"]  =dp[\"EvalDate\"] + timedelta(days=lag) \n    df = df.merge(dp, on=[\"playerId\", \"EvalDate\"], suffixes=[\"\",f\"_{lag}\"], how=\"left\")\n    return df\n#=================================\ndef test_lag(sub):\n    sub[\"playerId\"] = sub[\"date_playerId\"].apply(lambda s: int(  s.split(\"_\")[1]  ) )\n    assert sub.date.nunique() == 1\n    dte = sub[\"date\"].unique()[0]\n    \n    eval_dt = pd.to_datetime(dte, format=\"%Y%m%d\")\n    dtes = [eval_dt + timedelta(days=-k) for k in LAGS]\n    mp_dtes = {eval_dt + timedelta(days=-k):k for k in LAGS}\n    \n    sl = LAST.loc[LAST.EvalDate.between(dtes[-1], dtes[0]), [\"EvalDate\",\"playerId\"]+TGTCOLS].copy()\n    sl[\"EvalDate\"] = sl[\"EvalDate\"].map(mp_dtes)\n    du = [flatten(sl, col) for col in TGTCOLS]\n    du = reduce(reducer, du)\n    return du, eval_dt\n    #\n#===============\n\ntr = pd.read_csv(\"../input/mlb-data/target.csv\")\nprint(tr.shape)\ngc.collect()\n\ntr[\"EvalDate\"] = pd.to_datetime(tr[\"EvalDate\"])\ntr[\"EvalDate\"] = tr[\"EvalDate\"] + timedelta(days=-1)\ntr[\"EvalYear\"] = tr[\"EvalDate\"].dt.year\n\nMED_DF = tr.groupby([\"playerId\",\"EvalYear\"])[TGTCOLS].median().reset_index()\nMEDCOLS = [\"tgt1_med\",\"tgt2_med\", \"tgt3_med\", \"tgt4_med\"]\nMED_DF.columns = [\"playerId\",\"EvalYear\"] + MEDCOLS\n\nLAGS = list(range(1,21))\nFECOLS = [f\"{col}_{lag}\" for lag in reversed(LAGS) for col in TGTCOLS]\n\nfor lag in tqdm(LAGS):\n    tr = train_lag(tr, lag=lag)\n    gc.collect()\n#===========\ntr = tr.sort_values(by=[\"playerId\", \"EvalDate\"])\nprint(tr.shape)\ntr = tr.dropna()\nprint(tr.shape)\ntr = tr.merge(MED_DF, on=[\"playerId\",\"EvalYear\"])\ngc.collect()\n\nX = tr[FECOLS+MEDCOLS].values\ny = tr[TGTCOLS].values\ncl = tr[\"playerId\"].values\n\nNFOLDS = 6\nskf = StratifiedKFold(n_splits=NFOLDS)\nfolds = skf.split(X, cl)\nfolds = list(folds)\n\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\n\ntf.random.set_seed(777)\n\ndef make_model(n_in):\n    inp = L.Input(name=\"inputs\", shape=(n_in,))\n    x = L.Dense(50, activation=\"relu\", name=\"d1\")(inp)\n    x = L.Dense(50, activation=\"relu\", name=\"d2\")(x)\n    preds = L.Dense(4, activation=\"linear\", name=\"preds\")(x)\n    \n    model = M.Model(inp, preds, name=\"ANN\")\n    model.compile(loss=\"mean_absolute_error\", optimizer=\"adam\")\n    return model\n\nnet = make_model(X.shape[1])\nprint(net.summary())\n\noof = np.zeros(y.shape)\nnets = []\nfor idx in range(NFOLDS):\n    print(\"FOLD:\", idx)\n    tr_idx, val_idx = folds[idx]\n    ckpt = ModelCheckpoint(f\"w{idx}.h5\", monitor='val_loss', verbose=1, save_best_only=True,mode='min')\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2,patience=3, min_lr=0.0005)\n    es = EarlyStopping(monitor='val_loss', patience=6)\n    reg = make_model(X.shape[1])\n#     reg.fit(X[tr_idx], y[tr_idx], epochs=10, batch_size=35_000, validation_data=(X[val_idx], y[val_idx]),\n#             verbose=1, callbacks=[ckpt, reduce_lr, es])\n    reg.load_weights(f\"w{idx}.h5\")\n    oof[val_idx] = reg.predict(X[val_idx], batch_size=50_000, verbose=1)\n    nets.append(reg)\n    gc.collect()\n    #\n#\n\nmae = mean_absolute_error(y, oof)\nmse = mean_squared_error(y, oof, squared=False)\nprint(\"mae:\", mae)\nprint(\"mse:\", mse)\n\n# Historical information to use in prediction time\nbound_dt = pd.to_datetime(\"2021-01-01\")\nLAST = tr.loc[tr.EvalDate>bound_dt].copy()\n\nLAST_MED_DF = MED_DF.loc[MED_DF.EvalYear==2021].copy()\nLAST_MED_DF.drop(\"EvalYear\", axis=1, inplace=True)\ndel tr\n\n#\"\"\"\nimport mlb\nFE = []; SUB = [];","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:15:25.495162Z","iopub.status.idle":"2021-07-31T03:15:25.495763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"import copy\n\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in iter_test: # make predictions here\n    \n    sub = copy.deepcopy(sample_prediction_df.reset_index())\n    sample_prediction_df = copy.deepcopy(sample_prediction_df.reset_index(drop=True))\n    \n    # LGBM summit\n    # creat dataset\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    # Dealing with missing values\n    if test_df['rosters'].iloc[0] == test_df['rosters'].iloc[0]:\n        test_rosters = pd.DataFrame(eval(test_df['rosters'].iloc[0]))\n    else:\n        test_rosters = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in rosters.columns:\n            if col == 'playerId': continue\n            test_rosters[col] = np.nan\n            \n    if test_df['playerBoxScores'].iloc[0] == test_df['playerBoxScores'].iloc[0]:\n        test_scores = pd.DataFrame(eval(test_df['playerBoxScores'].iloc[0]))\n    else:\n        test_scores = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in scores.columns:\n            if col == 'playerId': continue\n            test_scores[col] = np.nan\n    test_scores = test_scores.groupby('playerId').sum().reset_index()\n    test = sample_prediction_df[['playerId']].copy()\n    test = test.merge(players[players_cols], on='playerId', how='left')\n    test = test.merge(test_rosters[rosters_cols], on='playerId', how='left')\n    test = test.merge(test_scores[scores_cols], on='playerId', how='left')\n    test = test.merge(player_target_stats, how='inner', left_on=[\"playerId\"],right_on=[\"playerId\"])\n    \n\n    test['label_playerId'] = test['playerId'].map(player2num)\n    test['label_primaryPositionName'] = test['primaryPositionName'].map(position2num)\n    test['label_teamId'] = test['teamId'].map(teamid2num)\n    test['label_status'] = test['status'].map(status2num)\n    \n    test_X = test[feature_cols]\n    # predict\n    pred1 = model1.predict(test_X)\n    \n    # predict\n    pred_lgd1 = model_lgb1.predict(test_X)\n    pred_lgd2 = model_lgb2.predict(test_X)\n    pred_lgd3 = model_lgb3.predict(test_X)\n    pred_lgd4 = model_lgb4.predict(test_X)\n    \n    pred_cat1 = model_cb1.predict(test_X)\n    pred_cat2 = model_cb2.predict(test_X)\n    pred_cat3 = model_cb3.predict(test_X)\n    pred_cat4 = model_cb4.predict(test_X)\n    \n    test['target1'] = np.clip(pred1,0,100)\n    test_X = test[feature_cols2]\n\n    pred2 = model2.predict(test_X)\n    pred3 = model3.predict(test_X)\n    pred4 = model4.predict(test_X)\n    \n    # merge submission\n    sample_prediction_df['target1'] = 1.00*np.clip(pred1, 0, 100)+0.00*np.clip(pred_lgd1, 0, 100)+0.00*np.clip(pred_cat1, 0, 100)\n    sample_prediction_df['target2'] = 0.10*np.clip(pred2, 0, 100)+0.65*np.clip(pred_lgd2, 0, 100)+0.25*np.clip(pred_cat2, 0, 100)\n    sample_prediction_df['target3'] = 0.65*np.clip(pred3, 0, 100)+0.25*np.clip(pred_lgd3, 0, 100)+0.10*np.clip(pred_cat3, 0, 100)\n    sample_prediction_df['target4'] = 0.65*np.clip(pred4, 0, 100)+0.25*np.clip(pred_lgd4, 0, 100)+0.10*np.clip(pred_cat4, 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)\n    del sample_prediction_df['playerId']\n    # TF summit\n    # Features computation at Evaluation Date\n    sub_fe, eval_dt = test_lag(sub)\n    sub_fe = sub_fe.merge(LAST_MED_DF, on=\"playerId\", how=\"left\")\n    sub_fe = sub_fe.fillna(0.)\n    \n    _preds = 0.\n    for reg in nets:\n        _preds += reg.predict(sub_fe[FECOLS + MEDCOLS]) / NFOLDS\n    sub_fe[TGTCOLS] = np.clip(_preds, 0, 100)\n    sub.drop([\"date\"]+TGTCOLS, axis=1, inplace=True)\n    sub = sub.merge(sub_fe[[\"playerId\"]+TGTCOLS], on=\"playerId\", how=\"left\")\n    sub.drop(\"playerId\", axis=1, inplace=True)\n    sub = sub.fillna(0.)\n    # Blending\n    blend = pd.concat(\n        [sub[['date_playerId']],\n        (0.35*sub.drop('date_playerId', axis=1) + 0.65*sample_prediction_df.drop('date_playerId', axis=1))],\n        axis=1\n    )\n    env.predict(blend)\n    # Update Available information\n    sub_fe[\"EvalDate\"] = eval_dt\n    #sub_fe.drop(MEDCOLS, axis=1, inplace=True)\n    LAST = LAST.append(sub_fe)\n    LAST = LAST.drop_duplicates(subset=[\"EvalDate\",\"playerId\"], keep=\"last\")","metadata":{"_uuid":"bc76106c-5572-4d75-ae68-28381f7406b7","_cell_guid":"c3148c3f-68b2-46f6-b8d6-3284e68b507a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:15:25.496736Z","iopub.status.idle":"2021-07-31T03:15:25.497392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat(\n    [sub[['date_playerId']],\n    (sub.drop('date_playerId', axis=1) + sample_prediction_df.drop('date_playerId', axis=1)) / 2],\n    axis=1\n)","metadata":{"_uuid":"9267b421-3cbb-4b28-a16e-e62ba8da58e4","_cell_guid":"cd1f93f4-37e0-47d0-b05a-0df03d07ce15","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:15:25.498303Z","iopub.status.idle":"2021-07-31T03:15:25.498904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction_df","metadata":{"_uuid":"99a89e72-692e-4aff-874b-a9e1d44ddaee","_cell_guid":"e94e45d6-2620-412b-bd5c-cb773241c8fe","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T03:15:25.499772Z","iopub.status.idle":"2021-07-31T03:15:25.500401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}}]}