{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"markdown","source":"## About Dataset","metadata":{"_uuid":"dedc4e57-daea-4f1f-9281-257352546646","_cell_guid":"7cda4356-b015-4992-877b-e40eb158b01e","trusted":true}},{"cell_type":"code","source":"!cp ../input/fork-of-1-35-lightgbm-ann-2505f2-c4e96a/* .","metadata":{"execution":{"iopub.status.busy":"2021-07-29T11:57:24.135307Z","iopub.execute_input":"2021-07-29T11:57:24.135901Z","iopub.status.idle":"2021-07-29T11:57:24.992554Z","shell.execute_reply.started":"2021-07-29T11:57:24.135812Z","shell.execute_reply":"2021-07-29T11:57:24.991305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2021-07-27T09:16:08.664328Z","iopub.execute_input":"2021-07-27T09:16:08.664702Z","iopub.status.idle":"2021-07-27T09:16:09.395745Z","shell.execute_reply.started":"2021-07-27T09:16:08.66467Z","shell.execute_reply":"2021-07-27T09:16:09.39485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\"\"\"\n!pip install pandarallel \n\nimport gc\n\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\nfrom pandarallel import pandarallel\npandarallel.initialize()\n\nBASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\ntrain = pd.read_csv(BASE_DIR / 'train.csv')\n\nnull = np.nan\ntrue = True\nfalse = False\n\nfor col in train.columns:\n\n    if col == 'date': continue\n\n    _index = train[col].notnull()\n    train.loc[_index, col] = train.loc[_index, col].parallel_apply(lambda x: eval(x))\n\n    outputs = []\n    for index, date, record in train.loc[_index, ['date', col]].itertuples():\n        _df = pd.DataFrame(record)\n        _df['index'] = index\n        _df['date'] = date\n        outputs.append(_df)\n\n    outputs = pd.concat(outputs).reset_index(drop=True)\n\n    outputs.to_csv(f'{col}_train.csv', index=False)\n    outputs.to_pickle(f'{col}_train.pkl')\n\n    del outputs\n    del train[col]\n    gc.collect()\n\"\"\"","metadata":{"_uuid":"62aca0d3-6af3-4760-9db0-0a397fdc5191","_cell_guid":"733fb600-c8da-4658-aa35-320c6817a8c5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-27T09:16:09.397426Z","iopub.execute_input":"2021-07-27T09:16:09.397868Z","iopub.status.idle":"2021-07-27T09:16:09.406466Z","shell.execute_reply.started":"2021-07-27T09:16:09.397831Z","shell.execute_reply":"2021-07-27T09:16:09.405765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{"_uuid":"ea4d85e1-21d7-4e7a-a8e7-36e5f53f4612","_cell_guid":"8d708767-e50d-4684-82a1-790feb5c0c1f","execution":{"iopub.status.busy":"2021-06-16T09:14:33.869464Z","iopub.execute_input":"2021-06-16T09:14:33.869905Z","iopub.status.idle":"2021-06-16T09:14:33.874766Z","shell.execute_reply.started":"2021-06-16T09:14:33.869879Z","shell.execute_reply":"2021-06-16T09:14:33.873097Z"},"trusted":true}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.metrics import mean_absolute_error\nfrom datetime import timedelta\nfrom functools import reduce\nfrom tqdm import tqdm\nimport lightgbm as lgbm\nimport mlb\nimport os\nimport pickle","metadata":{"_uuid":"2139878b-da24-41e3-bb59-76b60a1e16ef","_cell_guid":"3181642b-6cb0-424d-b5be-a0e6a5cb0457","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:21.54602Z","iopub.execute_input":"2021-07-31T07:49:21.546658Z","iopub.status.idle":"2021-07-31T07:49:23.89182Z","shell.execute_reply.started":"2021-07-31T07:49:21.546567Z","shell.execute_reply":"2021-07-31T07:49:23.890779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\nTRAIN_DIR = Path('../input/mlb-pdef-train-dataset')","metadata":{"_uuid":"3540a2ce-95e1-416f-9892-bcfa92cf6047","_cell_guid":"5f9dc680-5158-4bf2-856f-d43b6aa620de","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:23.894241Z","iopub.execute_input":"2021-07-31T07:49:23.894659Z","iopub.status.idle":"2021-07-31T07:49:23.899066Z","shell.execute_reply.started":"2021-07-31T07:49:23.894622Z","shell.execute_reply":"2021-07-31T07:49:23.898083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv(BASE_DIR / 'players.csv')\n\nrosters = pd.read_pickle(TRAIN_DIR / 'rosters_train.pkl')\ntargets = pd.read_pickle(TRAIN_DIR / 'nextDayPlayerEngagement_train.pkl')\nscores = pd.read_pickle(TRAIN_DIR / 'playerBoxScores_train.pkl')\nscores = scores.groupby(['playerId', 'date']).sum().reset_index()","metadata":{"_uuid":"38e056cf-cf8f-45f0-9ad4-cda6d11917a2","_cell_guid":"e67cda79-ae71-474a-861b-2b437b94a0a8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:23.900731Z","iopub.execute_input":"2021-07-31T07:49:23.901051Z","iopub.status.idle":"2021-07-31T07:49:28.47476Z","shell.execute_reply.started":"2021-07-31T07:49:23.901019Z","shell.execute_reply":"2021-07-31T07:49:28.473566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets_cols = ['playerId', 'target1', 'target2', 'target3', 'target4', 'date']\nplayers_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status', 'date']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'caughtStealing', 'stolenBases', 'atBats', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts',# 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'airOutsPitching',#lossesPitching', #'flyOutsPitching', ', ###\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves',#'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances', 'date']\n\nfeature_cols = ['label_playerId', #'label_primaryPositionName',# 'label_teamId', \n       'battingOrder', 'gamesPlayedBatting', 'flyOuts','label_status', #　'〇label_status',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'caughtStealing', 'stolenBases','atBats',  'groundIntoDoublePlay', #○'atBats', ' 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', #'sacBunts', #'sacFlies', 'catchersInterference', #'leftOnBase', '\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       #'lossesPitching', 'flyOutsPitching', 'airOutsPitching',###\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching',  'hitsPitching','intentionalWalksPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen',  'wildPitches', 'pickoffsPitching','balks',#○'pickoffsPitching', \n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching','saves','sacFliesPitching', #'holds', 'blownSaves',○'sacBuntsPitching', ○'sacFliesPitching' ○ 'sacFliesPitching'\n       'assists', 'putOuts', 'errors', 'chances','target1_mean',\n 'target1_median',\n 'target1_std',\n 'target1_min',\n 'target1_max',\n 'target1_prob',\n 'target2_mean',\n 'target2_median',\n 'target2_std',\n 'target2_min',\n 'target2_max',\n 'target2_prob',\n 'target3_mean',\n 'target3_median',\n 'target3_std',\n 'target3_min',\n 'target3_max',\n 'target3_prob',\n 'target4_mean',\n 'target4_median',\n 'target4_std',\n 'target4_min',\n 'target4_max',\n 'target4_prob']\n\nfeature_cols2 = ['label_playerId', 'label_primaryPositionName', 'label_teamId', #needed\n                 'label_status', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'hits', 'hitByPitch', 'intentionalWalks', #○'intentionalWalks'\n       'caughtStealing', 'stolenBases',  'groundIntoDoublePlay','atBats',#'〇atBats\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',#○ rbi \n       'leftOnBase','sacBunts',# 'sacFlies', 'catchersInterference',#', 〇leftOnBase'  \n       'gamesPlayedPitching', 'gamesStartedPitching','pickoffs',   #〇'pickoffs' \n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       #'airOutsPitching',# lossesPitching', #'flyOutsPitching', '',##〇\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'hitsPitching', 'intentionalWalksPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'wildPitches', 'pickoffsPitching',#'balks', \n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners', \n       'inheritedRunnersScored', #'catchersInterferencePitching', \n       'sacFliesPitching', 'saves','sacBuntsPitching', #'holds', 'blownSaves'   #〇sacBuntsPitching'\n       'assists', 'putOuts', 'errors', 'chances','target1_mean',\n 'target1_median',\n 'target1_std',\n 'target1_min',\n 'target1_max',\n 'target1_prob',\n 'target2_mean',\n 'target2_median',\n 'target2_std',\n 'target2_min',\n 'target2_max',\n 'target2_prob',\n 'target3_mean',\n 'target3_median',\n 'target3_std',\n 'target3_min',\n 'target3_max',\n 'target3_prob',\n 'target4_mean',\n 'target4_median',\n 'target4_std',\n 'target4_min',\n 'target4_max',\n 'target4_prob',\n    'target1']","metadata":{"_uuid":"046564c8-2d25-4540-9a94-9e629e263a22","_cell_guid":"39e17ba0-96e5-438c-94bd-6c184905c35f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:28.476316Z","iopub.execute_input":"2021-07-31T07:49:28.476635Z","iopub.status.idle":"2021-07-31T07:49:28.499074Z","shell.execute_reply.started":"2021-07-31T07:49:28.476605Z","shell.execute_reply":"2021-07-31T07:49:28.498153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_target_stats = pd.read_csv(\"../input/player-target-stats/player_target_stats.csv\")\ndata_names=player_target_stats.columns.values.tolist()\ndata_names","metadata":{"_uuid":"09b65c0a-4ac9-45cb-86ee-d81e871960a7","_cell_guid":"5ee50f59-b249-4698-ae86-e35289c6df01","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:28.50103Z","iopub.execute_input":"2021-07-31T07:49:28.50167Z","iopub.status.idle":"2021-07-31T07:49:28.550626Z","shell.execute_reply.started":"2021-07-31T07:49:28.501624Z","shell.execute_reply":"2021-07-31T07:49:28.549353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creat dataset\ntrain = targets[targets_cols].merge(players[players_cols], on=['playerId'], how='left')\ntrain = train.merge(rosters[rosters_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(scores[scores_cols], on=['playerId', 'date'], how='left')\ntrain = train.merge(player_target_stats, how='inner', left_on=[\"playerId\"],right_on=[\"playerId\"])\n\n\n# label encoding\nplayer2num = {c: i for i, c in enumerate(train['playerId'].unique())}\nposition2num = {c: i for i, c in enumerate(train['primaryPositionName'].unique())}\nteamid2num = {c: i for i, c in enumerate(train['teamId'].unique())}\nstatus2num = {c: i for i, c in enumerate(train['status'].unique())}\ntrain['label_playerId'] = train['playerId'].map(player2num)\ntrain['label_primaryPositionName'] = train['primaryPositionName'].map(position2num)\ntrain['label_teamId'] = train['teamId'].map(teamid2num)\ntrain['label_status'] = train['status'].map(status2num)","metadata":{"_uuid":"6cb3c4b2-a69b-4ce5-9d70-6633354c11ea","_cell_guid":"69624101-e45d-442a-bbb4-fb5514840e21","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:28.552523Z","iopub.execute_input":"2021-07-31T07:49:28.552943Z","iopub.status.idle":"2021-07-31T07:49:36.896457Z","shell.execute_reply.started":"2021-07-31T07:49:28.552896Z","shell.execute_reply":"2021-07-31T07:49:36.89542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[feature_cols]\ntrain_y = train[['target1', 'target2', 'target3', 'target4']]\n\n_index = (train['date'] < 20210401)\nx_train1 = train_X.loc[_index].reset_index(drop=True)\ny_train1 = train_y.loc[_index].reset_index(drop=True)\nx_valid1 = train_X.loc[~_index].reset_index(drop=True)\ny_valid1 = train_y.loc[~_index].reset_index(drop=True)","metadata":{"_uuid":"ddc79e19-ca6c-40d2-bb71-cf5df58549ce","_cell_guid":"ffc2b454-9e48-44f8-b4fa-d2d87ea5e89d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:36.897539Z","iopub.execute_input":"2021-07-31T07:49:36.897823Z","iopub.status.idle":"2021-07-31T07:49:40.931904Z","shell.execute_reply.started":"2021-07-31T07:49:36.897796Z","shell.execute_reply":"2021-07-31T07:49:40.930829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[feature_cols2]\ntrain_y = train[['target1', 'target2', 'target3', 'target4']]\n\n_index = (train['date'] < 20210401)\nx_train2 = train_X.loc[_index].reset_index(drop=True)\ny_train2 = train_y.loc[_index].reset_index(drop=True)\nx_valid2 = train_X.loc[~_index].reset_index(drop=True)\ny_valid2 = train_y.loc[~_index].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T07:49:40.933541Z","iopub.execute_input":"2021-07-31T07:49:40.933917Z","iopub.status.idle":"2021-07-31T07:49:43.391249Z","shell.execute_reply.started":"2021-07-31T07:49:40.933885Z","shell.execute_reply":"2021-07-31T07:49:43.390099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X","metadata":{"execution":{"iopub.status.busy":"2021-07-30T05:39:00.302963Z","iopub.execute_input":"2021-07-30T05:39:00.303456Z","iopub.status.idle":"2021-07-30T05:39:01.311375Z","shell.execute_reply.started":"2021-07-30T05:39:00.303426Z","shell.execute_reply":"2021-07-30T05:39:01.310152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_lgbm(x_train, y_train, x_valid, y_valid, params: dict=None, verbose=100):\n    oof_pred = np.zeros(len(y_valid), dtype=np.float32)\n    model = lgbm.LGBMRegressor(**params)\n    model.fit(x_train, y_train, \n        eval_set=[(x_valid, y_valid)],  \n        early_stopping_rounds=verbose, \n        verbose=verbose)\n    oof_pred = model.predict(x_valid)\n    score = mean_absolute_error(oof_pred, y_valid)\n    print('mae:', score)\n    return oof_pred, model, score\n\n\n# training lightgbm\n\nparams1 = {'objective':'mae',\n           'reg_alpha': 0.14947461820098767, \n           'reg_lambda': 0.10185644384043743, \n           'n_estimators': 3633, \n           'learning_rate': 0.08046301304430488, \n           'num_leaves': 674, \n           'feature_fraction': 0.9101240539122566, \n           'bagging_fraction': 0.9884451442950513, \n           'bagging_freq': 8, \n           'min_child_samples': 51}\n\nparams2 = {\n 'objective':'mae',\n 'reg_alpha': 0.1,\n 'reg_lambda': 0.1, \n 'n_estimators': 80,\n 'learning_rate': 0.1,\n 'random_state': 42,\n \"num_leaves\": 22\n}\n\nparams4 = {'objective':'mae',\n           'reg_alpha': 0.016468100279441976, \n           'reg_lambda': 0.09128335764019105, \n           'n_estimators': 9868, \n           'learning_rate': 0.10528150510326864, \n           'num_leaves': 157, \n           'feature_fraction': 0.5419185713426886, \n           'bagging_fraction': 0.2637405128936662, \n           'bagging_freq': 19, \n           'min_child_samples': 71}\n\n\nparams = {\n 'objective':'mae',\n 'reg_alpha': 0.1,\n 'reg_lambda': 0.1, \n 'n_estimators': 10000,\n 'learning_rate': 0.1,\n 'random_state': 42,\n \"num_leaves\": 100\n}\n\n\noof1, model1, score1 = fit_lgbm(\n    x_train1, y_train1['target1'],\n    x_valid1, y_valid1['target1'],\n    params1\n )\n\noof2, model2, score2 = fit_lgbm(\n    x_train2, y_train2['target2'],\n    x_valid2, y_valid2['target2'],\n    params2\n)\n\noof3, model3, score3 = fit_lgbm(\n    x_train2, y_train2['target3'],\n    x_valid2, y_valid2['target3'],\n   params\n)\n\noof4, model4, score4 = fit_lgbm(\n    x_train2, y_train2['target4'],\n    x_valid2, y_valid2['target4'],\n    params4\n)\n\nscore = (score1+score2+score3+score4) / 4\nprint(f'score: {score}')","metadata":{"_uuid":"c616b4b9-961c-4fcc-9643-391e8225a43c","_cell_guid":"50c7c4dd-9076-4e0a-98bc-bb2e56f0ecb4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-30T07:12:04.818426Z","iopub.execute_input":"2021-07-30T07:12:04.818819Z","iopub.status.idle":"2021-07-30T07:20:50.217045Z","shell.execute_reply.started":"2021-07-30T07:12:04.818783Z","shell.execute_reply":"2021-07-30T07:20:50.216202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nfrom catboost import CatBoostRegressor\n\ndef fit_lgbm(x_train, y_train, x_valid, y_valid, params: dict=None, verbose=100):\n    oof_pred_lgb = np.zeros(len(y_valid), dtype=np.float32)\n    oof_pred_cat = np.zeros(len(y_valid), dtype=np.float32)\n    \n#     if os.path.isfile(f'../input/mlb-lgbm-and-catboost-models/model_lgb_{target}.pkl'):\n#         with open(f'../input/mlb-lgbm-and-catboost-models/model_lgb_{target}.pkl', 'rb') as fin:\n#             model = pickle.load(fin)\n#     else:\n    \n    model = lgbm.LGBMRegressor(**params)\n    model.fit(x_train, y_train, \n        eval_set=[(x_valid, y_valid)],  \n        early_stopping_rounds=verbose, \n        verbose=verbose)\n\n#     with open(f'model_lgb_{target}.pkl', 'wb') as handle:\n#         pickle.dump(model, handle, protocol=pickle.HIGHEST_PROTOCOL)\n    \n    oof_pred_lgb = model.predict(x_valid)\n    score_lgb = mean_absolute_error(oof_pred_lgb, y_valid)\n    print('mae:', score_lgb)\n    \n#     if os.path.isfile(f'../input/mlb-lgbm-and-catboost-models/model_cb_{target}.pkl'):\n#         with open(f'../input/mlb-lgbm-and-catboost-models/model_cb_{target}.pkl', 'rb') as fin:\n#             model_cb = pickle.load(fin)\n#     else:\n    \n    model_cb = CatBoostRegressor(\n                n_estimators=2000,\n                learning_rate=0.05,\n                loss_function='MAE',\n                eval_metric='MAE',\n                max_bin=50,\n                subsample=0.9,\n                colsample_bylevel=0.5,\n                verbose=100)\n\n    model_cb.fit(x_train, y_train, use_best_model=True,\n                        eval_set=(x_valid, y_valid),\n                        early_stopping_rounds=25)\n\n#         with open(f'model_cb_{target}.pkl', 'wb') as handle:\n#             pickle.dump(model_cb, handle, protocol=pickle.HIGHEST_PROTOCOL)\n    \n    oof_pred_cat = model_cb.predict(x_valid)\n    score_cat = mean_absolute_error(oof_pred_cat, y_valid)\n    print('mae:', score_cat)\n    \n    return oof_pred_lgb, model, oof_pred_cat, model_cb, score_lgb, score_cat\n\n\n# training lightgbm\nparams = {\n'boosting_type': 'gbdt',\n'objective':'mae',\n'subsample': 0.5,\n'subsample_freq': 1,\n'learning_rate': 0.03,\n'num_leaves': 2**11-1,\n'min_data_in_leaf': 2**12-1,\n'feature_fraction': 0.5,\n'max_bin': 100,\n'n_estimators': 2500,\n'boost_from_average': False,\n\"random_seed\":42,\n}\n\noof_pred_lgb2, model_lgb2, oof_pred_cat2, model_cb2, score_lgb2, score_cat2 = fit_lgbm(\n    x_train1, y_train1['target2'],\n    x_valid1, y_valid1['target2'],\n    params\n)\n\noof_pred_lgb1, model_lgb1, oof_pred_cat1, model_cb1, score_lgb1, score_cat1 = fit_lgbm(\n    x_train1, y_train1['target1'],\n    x_valid1, y_valid1['target1'],\n    params\n)\n\noof_pred_lgb3, model_lgb3, oof_pred_cat3, model_cb3, score_lgb3, score_cat3 = fit_lgbm(\n    x_train1, y_train1['target3'],\n    x_valid1, y_valid1['target3'],\n    params\n)\noof_pred_lgb4, model_lgb4, oof_pred_cat4, model_cb4, score_lgb4, score_cat4= fit_lgbm(\n    x_train1, y_train1['target4'],\n    x_valid1, y_valid1['target4'],\n    params\n)\n\nscore = (score_lgb1+score_lgb2+score_lgb3+score_lgb4) / 4\nprint(f'LightGBM score: {score}')\n\nscore = (score_cat1+score_cat2+score_cat3+score_cat4) / 4\nprint(f'Catboost score: {score}')","metadata":{"execution":{"iopub.status.busy":"2021-07-27T09:59:46.308642Z","iopub.execute_input":"2021-07-27T09:59:46.309001Z","iopub.status.idle":"2021-07-27T10:09:32.889079Z","shell.execute_reply.started":"2021-07-27T09:59:46.308971Z","shell.execute_reply":"2021-07-27T10:09:32.887557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open('../input/modellgbcb/model_lgb_cb/model1.pickle', mode='rb') as f:\n#     model1 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model2.pickle', mode='rb') as f:\n#     model2 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model3.pickle', mode='rb') as f:\n#     model3 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model4.pickle', mode='rb') as f:\n#     model4 = pickle.load(f)\n    \n# with open('../input/modellgbcb/model_lgb_cb/model_lgb1.pickle', mode='rb') as f:\n#     model_lgb1 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model_lgb2.pickle', mode='rb') as f:\n#     model_lgb2 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model_lgb3.pickle', mode='rb') as f:\n#     model_lgb3 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model_lgb4.pickle', mode='rb') as f:\n#     model_lgb4 = pickle.load(f)\n    \n# with open('../input/modellgbcb/model_lgb_cb/model_cb1.pickle', mode='rb') as f:\n#     model_cb1 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model_cb2.pickle', mode='rb') as f:\n#     model_cb2 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model_cb3.pickle', mode='rb') as f:\n#     model_cb3 = pickle.load(f)\n# with open('../input/modellgbcb/model_lgb_cb/model_cb4.pickle', mode='rb') as f:\n#     model_cb4 = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T07:47:29.184709Z","iopub.execute_input":"2021-07-31T07:47:29.185123Z","iopub.status.idle":"2021-07-31T07:47:30.763777Z","shell.execute_reply.started":"2021-07-31T07:47:29.185087Z","shell.execute_reply":"2021-07-31T07:47:30.762557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'caughtStealing', 'stolenBases', 'atBats','groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts',# 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'airOutsPitching',#'lossesPitching', #'flyOutsPitching', '###\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', #'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances']\n\nnull = np.nan\ntrue = True\nfalse = False","metadata":{"_uuid":"7828a113-8dc7-4691-b2a6-b232ca0525dc","_cell_guid":"08cf9a6e-1f2c-4acd-b714-703d641d996a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:50.276732Z","iopub.execute_input":"2021-07-31T07:49:50.277088Z","iopub.status.idle":"2021-07-31T07:49:50.286063Z","shell.execute_reply.started":"2021-07-31T07:49:50.277057Z","shell.execute_reply":"2021-07-31T07:49:50.284718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datetime import timedelta\nfrom tqdm import tqdm\nimport gc\nfrom functools import reduce\nfrom sklearn.model_selection import StratifiedKFold\n\nROOT_DIR = \"../input/mlb-player-digital-engagement-forecasting\"\n\n#=======================#\ndef flatten(df, col):\n    du = (df.pivot(index=\"playerId\", columns=\"EvalDate\", \n               values=col).add_prefix(f\"{col}_\").\n      rename_axis(None, axis=1).reset_index())\n    return du\n#============================#\ndef reducer(left, right):\n    return left.merge(right, on=\"playerId\")\n#========================\n\nTGTCOLS = [\"target1\",\"target2\",\"target3\",\"target4\"]\ndef train_lag(df, lag=1):\n    dp = df[[\"playerId\",\"EvalDate\"]+TGTCOLS].copy()\n    dp[\"EvalDate\"]  =dp[\"EvalDate\"] + timedelta(days=lag) \n    df = df.merge(dp, on=[\"playerId\", \"EvalDate\"], suffixes=[\"\",f\"_{lag}\"], how=\"left\")\n    return df\n#=================================\ndef test_lag(sub):\n    sub[\"playerId\"] = sub[\"date_playerId\"].apply(lambda s: int(  s.split(\"_\")[1]  ) )\n    assert sub.date.nunique() == 1\n    dte = sub[\"date\"].unique()[0]\n    \n    eval_dt = pd.to_datetime(dte, format=\"%Y%m%d\")\n    dtes = [eval_dt + timedelta(days=-k) for k in LAGS]\n    mp_dtes = {eval_dt + timedelta(days=-k):k for k in LAGS}\n    \n    sl = LAST.loc[LAST.EvalDate.between(dtes[-1], dtes[0]), [\"EvalDate\",\"playerId\"]+TGTCOLS].copy()\n    sl[\"EvalDate\"] = sl[\"EvalDate\"].map(mp_dtes)\n    du = [flatten(sl, col) for col in TGTCOLS]\n    du = reduce(reducer, du)\n    return du, eval_dt\n    #\n#===============\n\ntr = pd.read_csv(\"../input/mlb-data/target.csv\")\nprint(tr.shape)\ngc.collect()\n\ntr[\"EvalDate\"] = pd.to_datetime(tr[\"EvalDate\"])\ntr[\"EvalDate\"] = tr[\"EvalDate\"] + timedelta(days=-1)\ntr[\"EvalYear\"] = tr[\"EvalDate\"].dt.year\n\nMED_DF = tr.groupby([\"playerId\",\"EvalYear\"])[TGTCOLS].median().reset_index()\nMEDCOLS = [\"tgt1_med\",\"tgt2_med\", \"tgt3_med\", \"tgt4_med\"]\nMED_DF.columns = [\"playerId\",\"EvalYear\"] + MEDCOLS\n\nLAGS = list(range(1,21))\nFECOLS = [f\"{col}_{lag}\" for lag in reversed(LAGS) for col in TGTCOLS]\n\nfor lag in tqdm(LAGS):\n    tr = train_lag(tr, lag=lag)\n    gc.collect()\n#===========\ntr = tr.sort_values(by=[\"playerId\", \"EvalDate\"])\nprint(tr.shape)\ntr = tr.dropna()\nprint(tr.shape)\ntr = tr.merge(MED_DF, on=[\"playerId\",\"EvalYear\"])\ngc.collect()\n\nX = tr[FECOLS+MEDCOLS].values\ny = tr[TGTCOLS].values\ncl = tr[\"playerId\"].values\n\nNFOLDS = 10\nskf = StratifiedKFold(n_splits=NFOLDS)\nfolds = skf.split(X, cl)\nfolds = list(folds)\n\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\n\ntf.random.set_seed(777)\n\ndef make_model(n_in):\n    inp = L.Input(name=\"inputs\", shape=(n_in,))\n    x = L.Dense(50, activation=\"relu\", name=\"d1\")(inp)\n    x = L.Dense(50, activation=\"relu\", name=\"d2\")(x)\n    preds = L.Dense(4, activation=\"linear\", name=\"preds\")(x)\n    \n    model = M.Model(inp, preds, name=\"ANN\")\n    model.compile(loss=\"mean_absolute_error\", optimizer=\"adam\")\n    return model\n\nnet = make_model(X.shape[1])\nprint(net.summary())\n\noof = np.zeros(y.shape)\nnets = []\nfor idx in range(NFOLDS):\n    print(\"FOLD:\", idx)\n    tr_idx, val_idx = folds[idx]\n    ckpt = ModelCheckpoint(f\"w{idx}.h5\", monitor='val_loss', verbose=1, save_best_only=True,mode='min')\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2,patience=3, min_lr=0.0005)\n    es = EarlyStopping(monitor='val_loss', patience=6)\n    reg = make_model(X.shape[1])\n    reg.fit(X[tr_idx], y[tr_idx], epochs=10, batch_size=35_000, validation_data=(X[val_idx], y[val_idx]),\n            verbose=1, callbacks=[ckpt, reduce_lr, es])\n    reg.load_weights(f\"w{idx}.h5\")\n    oof[val_idx] = reg.predict(X[val_idx], batch_size=50_000, verbose=1)\n    nets.append(reg)\n    gc.collect()\n\nmae = mean_absolute_error(y, oof)\nmse = mean_squared_error(y, oof, squared=False)\nprint(\"mae:\", mae)\nprint(\"mse:\", mse)\n\n# Historical information to use in prediction time\nbound_dt = pd.to_datetime(\"2021-01-01\")\nLAST = tr.loc[tr.EvalDate>bound_dt].copy()\n\nLAST_MED_DF = MED_DF.loc[MED_DF.EvalYear==2021].copy()\nLAST_MED_DF.drop(\"EvalYear\", axis=1, inplace=True)\ndel tr\n\n#\"\"\"\nimport mlb\nFE = []; SUB = [];","metadata":{"execution":{"iopub.status.busy":"2021-07-30T06:54:35.126221Z","iopub.execute_input":"2021-07-30T06:54:35.126869Z","iopub.status.idle":"2021-07-30T06:59:03.407972Z","shell.execute_reply.started":"2021-07-30T06:54:35.126773Z","shell.execute_reply":"2021-07-30T06:59:03.406867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"import copy\n\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in iter_test: # make predictions here\n    \n    sub = copy.deepcopy(sample_prediction_df.reset_index())\n    sample_prediction_df = copy.deepcopy(sample_prediction_df.reset_index(drop=True))\n    \n    # LGBM summit\n    # creat dataset\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    # Dealing with missing values\n    if test_df['rosters'].iloc[0] == test_df['rosters'].iloc[0]:\n        test_rosters = pd.DataFrame(eval(test_df['rosters'].iloc[0]))\n    else:\n        test_rosters = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in rosters.columns:\n            if col == 'playerId': continue\n            test_rosters[col] = np.nan\n            \n    if test_df['playerBoxScores'].iloc[0] == test_df['playerBoxScores'].iloc[0]:\n        test_scores = pd.DataFrame(eval(test_df['playerBoxScores'].iloc[0]))\n    else:\n        test_scores = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in scores.columns:\n            if col == 'playerId': continue\n            test_scores[col] = np.nan\n    test_scores = test_scores.groupby('playerId').sum().reset_index()\n    test = sample_prediction_df[['playerId']].copy()\n    test = test.merge(players[players_cols], on='playerId', how='left')\n    test = test.merge(test_rosters[rosters_cols], on='playerId', how='left')\n    test = test.merge(test_scores[scores_cols], on='playerId', how='left')\n    test = test.merge(player_target_stats, how='inner', left_on=[\"playerId\"],right_on=[\"playerId\"])\n    \n\n    test['label_playerId'] = test['playerId'].map(player2num)\n    test['label_primaryPositionName'] = test['primaryPositionName'].map(position2num)\n    test['label_teamId'] = test['teamId'].map(teamid2num)\n    test['label_status'] = test['status'].map(status2num)\n    \n    test_X = test[feature_cols]\n    \n    # predict\n    pred1 = model1.predict(test_X)\n    \n    pred_lgd1 = model_lgb1.predict(test_X)\n    pred_lgd2 = model_lgb2.predict(test_X)\n    pred_lgd3 = model_lgb3.predict(test_X)\n    pred_lgd4 = model_lgb4.predict(test_X)\n    \n    pred_cat1 = model_cb1.predict(test_X)\n    pred_cat2 = model_cb2.predict(test_X)\n    pred_cat3 = model_cb3.predict(test_X)\n    pred_cat4 = model_cb4.predict(test_X)\n    \n    test['target1'] = np.clip(pred1,0,100)\n    test_X = test[feature_cols2]\n    \n    # predict2\n    pred2 = model2.predict(test_X)\n    pred3 = model3.predict(test_X)\n    pred4 = model4.predict(test_X)\n    \n\n    \n    # merge submission\n    sample_prediction_df['target1'] = 1.00*np.clip(pred1, 0, 100)+0.00*np.clip(pred_lgd1, 0, 100)+0.00*np.clip(pred_cat1, 0, 100)\n    sample_prediction_df['target2'] = 0.10*np.clip(pred2, 0, 100)+0.65*np.clip(pred_lgd2, 0, 100)+0.25*np.clip(pred_cat2, 0, 100)\n    sample_prediction_df['target3'] = 0.65*np.clip(pred3, 0, 100)+0.25*np.clip(pred_lgd3, 0, 100)+0.10*np.clip(pred_cat3, 0, 100)\n    sample_prediction_df['target4'] = 0.65*np.clip(pred4, 0, 100)+0.25*np.clip(pred_lgd4, 0, 100)+0.10*np.clip(pred_cat4, 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)\n    del sample_prediction_df['playerId']\n    # TF summit\n    # Features computation at Evaluation Date\n    sub_fe, eval_dt = test_lag(sub)\n    sub_fe = sub_fe.merge(LAST_MED_DF, on=\"playerId\", how=\"left\")\n    sub_fe = sub_fe.fillna(0.)\n    \n    _preds = 0.\n    for reg in nets:\n        _preds += reg.predict(sub_fe[FECOLS + MEDCOLS]) / NFOLDS\n    sub_fe[TGTCOLS] = np.clip(_preds, 0, 100)\n    sub.drop([\"date\"]+TGTCOLS, axis=1, inplace=True)\n    sub = sub.merge(sub_fe[[\"playerId\"]+TGTCOLS], on=\"playerId\", how=\"left\")\n    sub.drop(\"playerId\", axis=1, inplace=True)\n    sub = sub.fillna(0.)\n    # Blending\n    blend = pd.concat(\n        [sub[['date_playerId']],\n        (0.35*sub.drop('date_playerId', axis=1) + 0.65*sample_prediction_df.drop('date_playerId', axis=1))],\n        axis=1\n    )\n    env.predict(blend)\n    # Update Available information\n    sub_fe[\"EvalDate\"] = eval_dt\n    #sub_fe.drop(MEDCOLS, axis=1, inplace=True)\n    LAST = LAST.append(sub_fe)\n    LAST = LAST.drop_duplicates(subset=[\"EvalDate\",\"playerId\"], keep=\"last\")","metadata":{"_uuid":"bc76106c-5572-4d75-ae68-28381f7406b7","_cell_guid":"c3148c3f-68b2-46f6-b8d6-3284e68b507a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-07-31T07:49:53.430987Z","iopub.execute_input":"2021-07-31T07:49:53.431357Z","iopub.status.idle":"2021-07-31T07:49:54.636363Z","shell.execute_reply.started":"2021-07-31T07:49:53.431313Z","shell.execute_reply":"2021-07-31T07:49:54.633466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat(\n    [sub[['date_playerId']],\n    (sub.drop('date_playerId', axis=1) + sample_prediction_df.drop('date_playerId', axis=1)) / 2],\n    axis=1\n)","metadata":{"_uuid":"9267b421-3cbb-4b28-a16e-e62ba8da58e4","_cell_guid":"cd1f93f4-37e0-47d0-b05a-0df03d07ce15","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-26T07:17:02.264951Z","iopub.status.idle":"2021-06-26T07:17:02.265581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction_df","metadata":{"_uuid":"99a89e72-692e-4aff-874b-a9e1d44ddaee","_cell_guid":"e94e45d6-2620-412b-bd5c-cb773241c8fe","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2021-06-26T07:17:02.26657Z","iopub.status.idle":"2021-06-26T07:17:02.267169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}}]}