{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datetime import timedelta\nfrom tqdm import tqdm\nimport gc\nfrom functools import reduce\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-18T12:34:46.762401Z","iopub.execute_input":"2021-06-18T12:34:46.763131Z","iopub.status.idle":"2021-06-18T12:34:47.756920Z","shell.execute_reply.started":"2021-06-18T12:34:46.763034Z","shell.execute_reply":"2021-06-18T12:34:47.755663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_df(df, col, bool_in=False):\n    tp = df.loc[ ~df[col].isnull() ,[col]].copy()\n    df.drop(col, axis=1, inplace=True)\n    \n    tp[col] = tp[col].str.replace(\"null\",'\"\"')\n    if bool_in:\n        tp[col] = tp[col].str.replace(\"false\",'\"False\"')\n        tp[col] = tp[col].str.replace(\"true\",'\"True\"')\n    tp[col] = tp[col].apply(lambda x: eval(x) )\n    a = tp[col].sum()\n    gc.collect()\n    return pd.DataFrame(a)\n#===============","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:47.758554Z","iopub.execute_input":"2021-06-18T12:34:47.758872Z","iopub.status.idle":"2021-06-18T12:34:47.767406Z","shell.execute_reply.started":"2021-06-18T12:34:47.758838Z","shell.execute_reply":"2021-06-18T12:34:47.765912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = \"../input/mlb-player-digital-engagement-forecasting\"","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:47.769450Z","iopub.execute_input":"2021-06-18T12:34:47.769768Z","iopub.status.idle":"2021-06-18T12:34:47.779617Z","shell.execute_reply.started":"2021-06-18T12:34:47.769733Z","shell.execute_reply":"2021-06-18T12:34:47.778520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_seed = 1998\nnp.random.seed(my_seed)\nimport random \nrandom.seed(my_seed)\nimport tensorflow as tf\ntf.random.set_seed(my_seed)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:47.781419Z","iopub.execute_input":"2021-06-18T12:34:47.781747Z","iopub.status.idle":"2021-06-18T12:34:53.996761Z","shell.execute_reply.started":"2021-06-18T12:34:47.781715Z","shell.execute_reply":"2021-06-18T12:34:53.995601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## UTILITY FUNCTIONS","metadata":{}},{"cell_type":"code","source":"#=======================#\ndef flatten(df, col):\n    du = (df.pivot(index=\"playerId\", columns=\"EvalDate\", \n               values=col).add_prefix(f\"{col}_\").\n      rename_axis(None, axis=1).reset_index())\n    return du\n#============================#\ndef reducer(left, right):\n    return left.merge(right, on=\"playerId\")\n#========================","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:53.998315Z","iopub.execute_input":"2021-06-18T12:34:53.998743Z","iopub.status.idle":"2021-06-18T12:34:54.005487Z","shell.execute_reply.started":"2021-06-18T12:34:53.998699Z","shell.execute_reply":"2021-06-18T12:34:54.004297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TGTCOLS = [\"target1\",\"target2\",\"target3\",\"target4\"]\ndef train_lag(df, lag=1):\n    dp = df[[\"playerId\",\"EvalDate\"]+TGTCOLS].copy()\n    dp[\"EvalDate\"]  =dp[\"EvalDate\"] + timedelta(days=lag) \n    df = df.merge(dp, on=[\"playerId\", \"EvalDate\"], suffixes=[\"\",f\"_{lag}\"], how=\"left\")\n    return df\n#=================================\ndef test_lag(sub):\n    sub[\"playerId\"] = sub[\"date_playerId\"].apply(lambda s: int(  s.split(\"_\")[1]  ) )\n    assert sub.date.nunique() == 1\n    dte = sub[\"date\"].unique()[0]\n    \n    eval_dt = pd.to_datetime(dte, format=\"%Y%m%d\")\n    dtes = [eval_dt + timedelta(days=-k) for k in LAGS]\n    mp_dtes = {eval_dt + timedelta(days=-k):k for k in LAGS}\n    \n    sl = LAST.loc[LAST.EvalDate.between(dtes[-1], dtes[0]), [\"EvalDate\",\"playerId\"]+TGTCOLS].copy()\n    sl[\"EvalDate\"] = sl[\"EvalDate\"].map(mp_dtes)\n    du = [flatten(sl, col) for col in TGTCOLS]\n    du = reduce(reducer, du)\n    return du, eval_dt\n    #\n#===============","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:54.007085Z","iopub.execute_input":"2021-06-18T12:34:54.007721Z","iopub.status.idle":"2021-06-18T12:34:54.023011Z","shell.execute_reply.started":"2021-06-18T12:34:54.007673Z","shell.execute_reply":"2021-06-18T12:34:54.022093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#tr = pd.read_csv(f\"{ROOT_DIR}/train.csv\")\ntr = pd.read_csv(\"../input/mlb-data/target.csv\")\nprint(tr.shape)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:54.024453Z","iopub.execute_input":"2021-06-18T12:34:54.025157Z","iopub.status.idle":"2021-06-18T12:34:58.520352Z","shell.execute_reply.started":"2021-06-18T12:34:54.025107Z","shell.execute_reply":"2021-06-18T12:34:58.519183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr[\"EvalDate\"] = pd.to_datetime(tr[\"EvalDate\"])\ntr[\"EvalDate\"] = tr[\"EvalDate\"] + timedelta(days=-1)\ntr[\"EvalYear\"] = tr[\"EvalDate\"].dt.year\ntr[\"EvalMonth\"] = tr[\"EvalDate\"].dt.month\ntr[\"EvalWeek\"] = tr[\"EvalDate\"].dt.weekday","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:34:58.522720Z","iopub.execute_input":"2021-06-18T12:34:58.523018Z","iopub.status.idle":"2021-06-18T12:34:59.721434Z","shell.execute_reply.started":"2021-06-18T12:34:58.522989Z","shell.execute_reply":"2021-06-18T12:34:59.720337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MED_DF = tr.groupby([\"playerId\",\"EvalYear\",\"EvalMonth\",\"EvalWeek\"])[TGTCOLS].median().reset_index()\nMEDCOLS = [\"tgt1_med\",\"tgt2_med\", \"tgt3_med\", \"tgt4_med\"]\nMED_DF.columns = [\"playerId\",\"EvalYear\",\"EvalMonth\",\"EvalWeek\"] + MEDCOLS\n","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:35:36.022616Z","iopub.execute_input":"2021-06-18T12:35:36.023107Z","iopub.status.idle":"2021-06-18T12:35:36.898774Z","shell.execute_reply.started":"2021-06-18T12:35:36.023060Z","shell.execute_reply":"2021-06-18T12:35:36.897816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MED_DF.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:35:38.948524Z","iopub.execute_input":"2021-06-18T12:35:38.948991Z","iopub.status.idle":"2021-06-18T12:35:38.975806Z","shell.execute_reply.started":"2021-06-18T12:35:38.948951Z","shell.execute_reply":"2021-06-18T12:35:38.974075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LAGS = list(range(1,21))\nFECOLS = [f\"{col}_{lag}\" for lag in reversed(LAGS) for col in TGTCOLS]","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:35:46.784291Z","iopub.execute_input":"2021-06-18T12:35:46.784694Z","iopub.status.idle":"2021-06-18T12:35:46.790660Z","shell.execute_reply.started":"2021-06-18T12:35:46.784657Z","shell.execute_reply":"2021-06-18T12:35:46.789020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LAGS","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:35:50.608479Z","iopub.execute_input":"2021-06-18T12:35:50.608864Z","iopub.status.idle":"2021-06-18T12:35:50.616137Z","shell.execute_reply.started":"2021-06-18T12:35:50.608828Z","shell.execute_reply":"2021-06-18T12:35:50.614920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor lag in tqdm(LAGS):\n    tr = train_lag(tr, lag=lag)\n    gc.collect()\n#===========\ntr = tr.sort_values(by=[\"playerId\", \"EvalDate\"])\nprint(tr.shape)\ntr = tr.dropna()\nprint(tr.shape)\ntr = tr.merge(MED_DF, on=[\"playerId\",\"EvalYear\",\"EvalMonth\",\"EvalWeek\"])\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:35:55.529608Z","iopub.execute_input":"2021-06-18T12:35:55.530054Z","iopub.status.idle":"2021-06-18T12:37:02.219285Z","shell.execute_reply.started":"2021-06-18T12:35:55.530014Z","shell.execute_reply":"2021-06-18T12:37:02.218190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:37:04.005678Z","iopub.execute_input":"2021-06-18T12:37:04.006153Z","iopub.status.idle":"2021-06-18T12:37:04.036518Z","shell.execute_reply.started":"2021-06-18T12:37:04.006107Z","shell.execute_reply":"2021-06-18T12:37:04.035098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:37:14.171486Z","iopub.execute_input":"2021-06-18T12:37:14.171864Z","iopub.status.idle":"2021-06-18T12:37:14.204484Z","shell.execute_reply.started":"2021-06-18T12:37:14.171825Z","shell.execute_reply":"2021-06-18T12:37:14.203725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TIMECOLS = ['EvalYear','EvalMonth','EvalWeek']","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:37:20.151269Z","iopub.execute_input":"2021-06-18T12:37:20.151818Z","iopub.status.idle":"2021-06-18T12:37:20.156513Z","shell.execute_reply.started":"2021-06-18T12:37:20.151763Z","shell.execute_reply":"2021-06-18T12:37:20.155290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = tr[FECOLS+MEDCOLS].values\ny = tr[TGTCOLS].values\ncl = tr[\"playerId\"].values","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:40:47.012262Z","iopub.execute_input":"2021-06-18T12:40:47.012687Z","iopub.status.idle":"2021-06-18T12:40:47.543692Z","shell.execute_reply.started":"2021-06-18T12:40:47.012638Z","shell.execute_reply":"2021-06-18T12:40:47.542749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NFOLDS = 2\nskf = StratifiedKFold(n_splits=NFOLDS)\nfolds = skf.split(X, cl)\nfolds = list(folds)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:40:58.181243Z","iopub.execute_input":"2021-06-18T12:40:58.181764Z","iopub.status.idle":"2021-06-18T12:41:02.953245Z","shell.execute_reply.started":"2021-06-18T12:40:58.181706Z","shell.execute_reply":"2021-06-18T12:41:02.952205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:41:04.537952Z","iopub.execute_input":"2021-06-18T12:41:04.538373Z","iopub.status.idle":"2021-06-18T12:41:04.545607Z","shell.execute_reply.started":"2021-06-18T12:41:04.538335Z","shell.execute_reply":"2021-06-18T12:41:04.544377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Neural Net Training","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:37:43.542541Z","iopub.execute_input":"2021-06-18T12:37:43.543045Z","iopub.status.idle":"2021-06-18T12:37:43.548575Z","shell.execute_reply.started":"2021-06-18T12:37:43.543012Z","shell.execute_reply":"2021-06-18T12:37:43.547425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model(n_in):\n    inp = L.Input(name=\"inputs\", shape=(n_in,))\n    x = L.Dense(64, activation=\"relu\", name=\"d1\")(inp)\n    x = L.Dense(64, activation=\"relu\", name=\"d2\")(x)\n    preds = L.Dense(4, activation=\"linear\", name=\"preds\")(x)\n    \n    model = M.Model(inp, preds, name=\"MLP\")\n    model.compile(loss=\"mean_absolute_error\", optimizer=\"adam\")\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:41:10.508044Z","iopub.execute_input":"2021-06-18T12:41:10.508472Z","iopub.status.idle":"2021-06-18T12:41:10.515431Z","shell.execute_reply.started":"2021-06-18T12:41:10.508436Z","shell.execute_reply":"2021-06-18T12:41:10.514400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net = make_model(X.shape[1])\nprint(net.summary())","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:41:16.214725Z","iopub.execute_input":"2021-06-18T12:41:16.215306Z","iopub.status.idle":"2021-06-18T12:41:16.259777Z","shell.execute_reply.started":"2021-06-18T12:41:16.215259Z","shell.execute_reply":"2021-06-18T12:41:16.258802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = np.zeros(y.shape)\nnets = []\nfor idx in range(NFOLDS):\n    print(\"FOLD:\", idx)\n    tr_idx, val_idx = folds[idx]\n    ckpt = ModelCheckpoint(f\"w{idx}.h5\", monitor='val_loss', verbose=1, save_best_only=True,mode='min')\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2,patience=3, min_lr=0.0001)\n    es = EarlyStopping(monitor='val_loss', patience=5)\n    reg = make_model(X.shape[1])\n    reg.fit(X[tr_idx], y[tr_idx], epochs=10, batch_size=30_000, validation_data=(X[val_idx], y[val_idx]),\n            verbose=1, callbacks=[ckpt, reduce_lr, es])\n    reg.load_weights(f\"w{idx}.h5\")\n    oof[val_idx] = reg.predict(X[val_idx], batch_size=50_000, verbose=1)\n    nets.append(reg)\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:41:23.476583Z","iopub.execute_input":"2021-06-18T12:41:23.477020Z","iopub.status.idle":"2021-06-18T12:42:10.298270Z","shell.execute_reply.started":"2021-06-18T12:41:23.476983Z","shell.execute_reply":"2021-06-18T12:42:10.297441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mae = mean_absolute_error(y, oof)\nmse = mean_squared_error(y, oof, squared=False)\nprint(\"mae:\", mae)\nprint(\"mse:\", mse)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:42:14.414064Z","iopub.execute_input":"2021-06-18T12:42:14.414566Z","iopub.status.idle":"2021-06-18T12:42:14.651656Z","shell.execute_reply.started":"2021-06-18T12:42:14.414532Z","shell.execute_reply":"2021-06-18T12:42:14.650413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Historical information to use in prediction time\nbound_dt = pd.to_datetime(\"2021-01-01\")\nLAST = tr.loc[tr.EvalDate>bound_dt].copy()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:42:23.044746Z","iopub.execute_input":"2021-06-18T12:42:23.045218Z","iopub.status.idle":"2021-06-18T12:42:23.322333Z","shell.execute_reply.started":"2021-06-18T12:42:23.045177Z","shell.execute_reply":"2021-06-18T12:42:23.320823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LAST_MED_DF = MED_DF.loc[MED_DF.EvalYear==2021].copy()\n\nLAST_MED_DF.drop(\"EvalYear\", axis=1, inplace=True)\ndel tr","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:42:28.290646Z","iopub.execute_input":"2021-06-18T12:42:28.291249Z","iopub.status.idle":"2021-06-18T12:42:28.329946Z","shell.execute_reply.started":"2021-06-18T12:42:28.291207Z","shell.execute_reply":"2021-06-18T12:42:28.329016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LAST.shape, LAST_MED_DF.shape, MED_DF.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:42:32.664761Z","iopub.execute_input":"2021-06-18T12:42:32.665490Z","iopub.status.idle":"2021-06-18T12:42:32.671399Z","shell.execute_reply.started":"2021-06-18T12:42:32.665443Z","shell.execute_reply":"2021-06-18T12:42:32.670542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb\nFE = []; SUB = [];\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sub) in iter_test:\n    # Features computation at Evaluation Date\n    sub = sub.reset_index()\n    sub_fe, eval_dt = test_lag(sub)\n    sub_fe = sub_fe.merge(LAST_MED_DF, on=\"playerId\", how=\"left\")\n    sub_fe = sub_fe.fillna(0.)\n    \n    _preds = 0.\n    for reg in nets:\n        _preds += reg.predict(sub_fe[FECOLS + MEDCOLS]) / NFOLDS\n    sub_fe[TGTCOLS] = np.clip(_preds, 0, 100)\n    sub.drop([\"date\"]+TGTCOLS, axis=1, inplace=True)\n    sub = sub.merge(sub_fe[[\"playerId\"]+TGTCOLS], on=\"playerId\", how=\"left\")\n    sub.drop(\"playerId\", axis=1, inplace=True)\n    sub = sub.fillna(0.)\n    sub = sub.drop_duplicates(subset=['date_playerId'], keep=\"first\")\n    # Submit\n    env.predict(sub)\n    # Update Available information\n    sub_fe[\"EvalDate\"] = eval_dt\n    #sub_fe.drop(MEDCOLS, axis=1, inplace=True)\n    LAST = LAST.append(sub_fe)\n    LAST = LAST.drop_duplicates(subset=[\"EvalDate\",\"playerId\"], keep=\"last\")","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:42:43.230088Z","iopub.execute_input":"2021-06-18T12:42:43.230831Z","iopub.status.idle":"2021-06-18T12:43:05.707440Z","shell.execute_reply.started":"2021-06-18T12:42:43.230783Z","shell.execute_reply":"2021-06-18T12:43:05.706426Z"},"trusted":true},"execution_count":null,"outputs":[]}]}