{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import re\nimport pickle","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:53.259129Z","iopub.execute_input":"2021-07-27T05:04:53.259575Z","iopub.status.idle":"2021-07-27T05:04:53.270571Z","shell.execute_reply.started":"2021-07-27T05:04:53.259511Z","shell.execute_reply":"2021-07-27T05:04:53.269574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tqdm\nimport seaborn as sns\nfrom sklearn.multioutput import MultiOutputRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-27T05:04:53.275472Z","iopub.execute_input":"2021-07-27T05:04:53.275792Z","iopub.status.idle":"2021-07-27T05:04:54.506210Z","shell.execute_reply.started":"2021-07-27T05:04:53.275761Z","shell.execute_reply":"2021-07-27T05:04:54.505151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.507953Z","iopub.execute_input":"2021-07-27T05:04:54.508304Z","iopub.status.idle":"2021-07-27T05:04:54.519972Z","shell.execute_reply.started":"2021-07-27T05:04:54.508273Z","shell.execute_reply":"2021-07-27T05:04:54.518966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_COLS = [\"target1\", \"target2\", \"target3\", \"target4\"]\nROLLING_WINDOWS = [7, 30]\nCATEGORICAL_FEATURES = [\n    \"home\",\n    \"positionCode\",\n    \"birthCountry\",\n    \"day_of_week\",\n    \"week_of_year\",\n    \"month\",\n    \"toTeamId\",\n    \"fromTeamId\",\n    \"teamId\",\n    \"statusCode\",\n    \"leagueId\",\n    \"divisionId\"\n]\nBOX_SCORES_COMPARISON_COLS = [\n    \"flyOuts\",\n    \"runsScored\",\n    \"doubles\",\n    \"triples\",\n    \"homeRuns\",\n    \"strikeOuts\",\n    \"baseOnBalls\",\n    \"intentionalWalks\",\n    \"hits\",\n    \"stolenBases\",\n    \"plateAppearances\",\n    \"totalBases\",\n    \"rbi\",\n    \"errors\",\n    \"chances\",\n]\nBOX_SCORES_CATEGORICAL_COLS = [\"home\", \"positionCode\"]\nBOX_SCORES_NUMERIC_COLS = [\"battingOrder\"]\nPLAYERS_COLS = [\"playerId\", \"DOB\", \"mlbDebutDate\", \"birthCountry\", \"salary\"]\nSEASONS_DAYS_FROM_COLS = [\n    \"seasonStartDate\",\n    \"preSeasonStartDate\",\n    \"regularSeasonStartDate\",\n    \"postSeasonStartDate\",\n    \"postSeasonEndDate\",\n]\nSEASONS_DATE_COLS = [\n    \"allStarDate\",\n    \"lastDate1stHalf\",\n    \"firstDate2ndHalf\",\n    \"seasonEndDate\",\n    \"preSeasonEndDate\",\n    \"regularSeasonEndDate\",\n    \"postSeasonEndDate\",\n]\nTRANSACTIONS_COLS = [\n    \"toTeamId\",\n    \"fromTeamId\",\n    \"SFA\",\n    \"TR\",\n    \"NUM\",\n    \"ASG\",\n    \"DES\",\n    \"CLW\",\n    \"OUT\",\n    \"REL\",\n    \"SC\",\n    \"OPT\",\n    \"RTN\",\n    \"SGN\",\n    \"SE\",\n    \"CU\",\n    \"DFA\",\n    \"RET\",\n]\nROSTERS_COLS = [\"date\", \"playerId\", \"teamId\", \"statusCode\"]\nPLAYER_TWITTER_COLS = [\"year\", \"month\", \"playerId\", \"player_followers_depth\"]\nTEAM_TWITTER_COLS = [\"year\", \"month\", \"teamId\", \"team_followers_depth\"]\nTEAMS_COLS = [\"teamId\", \"leagueId\", \"divisionId\"]\nAWARDS_COLS = [\"playerId\", \"awardsCount\", \"allStarCount\", \"mvpCount\", \"rookieCount\"]","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.521477Z","iopub.execute_input":"2021-07-27T05:04:54.521825Z","iopub.status.idle":"2021-07-27T05:04:54.533715Z","shell.execute_reply.started":"2021-07-27T05:04:54.521793Z","shell.execute_reply":"2021-07-27T05:04:54.532730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES_COLS = ['target1_mean', 'target1_max', 'target1_std', 'target2_mean',\n       'target2_max', 'target2_std', 'target3_mean', 'target3_max',\n       'target3_std', 'target4_mean', 'target4_max', 'target4_std',\n       'day_of_week', 'week_of_year', 'month', 'relative_flyOuts',\n       'relative_runsScored', 'relative_doubles', 'relative_triples',\n       'relative_homeRuns', 'relative_strikeOuts', 'relative_baseOnBalls',\n       'relative_intentionalWalks', 'relative_hits', 'relative_stolenBases',\n       'relative_plateAppearances', 'relative_totalBases', 'relative_rbi',\n       'relative_errors', 'relative_chances', 'battingOrder', 'home',\n       'positionCode', 'birthCountry', 'salary', 'age', 'debut_days',\n       'days_from_seasonStartDate', 'days_from_preSeasonStartDate',\n       'days_from_regularSeasonStartDate', 'days_from_postSeasonStartDate',\n       'days_from_postSeasonEndDate', 'is_allStarDate', 'is_lastDate1stHalf',\n       'is_firstDate2ndHalf', 'is_seasonEndDate', 'is_preSeasonEndDate',\n       'is_regularSeasonEndDate', 'is_postSeasonEndDate', 'is_restSeason',\n       'historicalAwardsCount', 'historicalAllStarCount', 'historicalMvpCount',\n       'historicalRookieCount', 'awardsCount', 'allStarCount', 'mvpCount',\n       'rookieCount', 'fromTeamId', 'toTeamId', 'ASG', 'CLW', 'CU', 'DES',\n       'DFA', 'NUM', 'OPT', 'OUT', 'REL', 'RET', 'RTN', 'SC', 'SE', 'SFA',\n       'SGN', 'TR', 'teamId', 'statusCode', 'player_followers_depth',\n       'team_followers_depth', 'leagueId', 'divisionId']","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.535177Z","iopub.execute_input":"2021-07-27T05:04:54.535517Z","iopub.status.idle":"2021-07-27T05:04:54.549714Z","shell.execute_reply.started":"2021-07-27T05:04:54.535485Z","shell.execute_reply":"2021-07-27T05:04:54.548608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = \"../input/mlb-player-digital-engagement-forecasting\"\nCUSTOM_DATA_ROOT = \"../input/datav7fixed\"","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.551102Z","iopub.execute_input":"2021-07-27T05:04:54.551463Z","iopub.status.idle":"2021-07-27T05:04:54.567104Z","shell.execute_reply.started":"2021-07-27T05:04:54.551433Z","shell.execute_reply":"2021-07-27T05:04:54.566128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unpack_col_to_df(col):\n    try:\n        output = pd.concat(\n            [pd.read_json(row) for row in col if isinstance(row, str)], ignore_index=True\n        )\n        date_cols = [col for col in output.columns if re.search(\"date\", col.lower())]\n        output[date_cols] = (\n            pd.concat([pd.to_datetime(output[col]) for col in date_cols], axis=1)\n            if len(date_cols) > 0\n            else output[date_cols]\n        )\n    except (TypeError, ValueError):\n        return pd.DataFrame()\n    \n    return output","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.568648Z","iopub.execute_input":"2021-07-27T05:04:54.568966Z","iopub.status.idle":"2021-07-27T05:04:54.581348Z","shell.execute_reply.started":"2021-07-27T05:04:54.568936Z","shell.execute_reply":"2021-07-27T05:04:54.580365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=False):\n    df = df.copy()\n    numerics = [\"int16\", \"int32\", \"int64\", \"float16\", \"float32\", \"float64\"]\n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == \"int\":\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if (\n                    c_min > np.finfo(np.float16).min\n                    and c_max < np.finfo(np.float16).max\n                ):\n                    df[col] = df[col].astype(np.float16)\n                elif (\n                    c_min > np.finfo(np.float32).min\n                    and c_max < np.finfo(np.float32).max\n                ):\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024 ** 2\n    if verbose:\n        print(\n            \"Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)\".format(\n                end_mem, 100 * (start_mem - end_mem) / start_mem\n            )\n        )\n    return df","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2021-07-27T05:04:54.584233Z","iopub.execute_input":"2021-07-27T05:04:54.584567Z","iopub.status.idle":"2021-07-27T05:04:54.599893Z","shell.execute_reply.started":"2021-07-27T05:04:54.584526Z","shell.execute_reply":"2021-07-27T05:04:54.599079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_players(df):\n    df = df.copy()\n    \n    df[\"age\"] = (df[\"date\"] - df[\"DOB\"]).dt.days\n    df[\"debut_days\"] = (df[\"date\"] - df[\"mlbDebutDate\"]).dt.days\n    return df.drop(columns=[\"DOB\", \"mlbDebutDate\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.601444Z","iopub.execute_input":"2021-07-27T05:04:54.601730Z","iopub.status.idle":"2021-07-27T05:04:54.617584Z","shell.execute_reply.started":"2021-07-27T05:04:54.601703Z","shell.execute_reply":"2021-07-27T05:04:54.616513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_box_scores(df):\n    df = df.drop(columns=[\"gamePk\", \"teamId\", \"teamName\", \"playerName\", \"gameTimeUTC\", \"jerseyNum\"])\n    df = df.rename(columns={\"gameDate\": \"date\"})\n    df.date = pd.to_datetime(df.date)\n    df = reduce_mem_usage(df)\n    output = df.groupby([\"date\", \"playerId\"], as_index=False).mean()\n    \n    return output","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.621088Z","iopub.execute_input":"2021-07-27T05:04:54.621434Z","iopub.status.idle":"2021-07-27T05:04:54.633412Z","shell.execute_reply.started":"2021-07-27T05:04:54.621401Z","shell.execute_reply":"2021-07-27T05:04:54.632184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_transactions(df):\n    df = df.copy()\n    df = df.drop(\n        columns=[\"transactionId\", \"description\"]\n    ).drop_duplicates()\n\n    df[\"constant\"] = 1\n\n    df[[\"fromTeamId\", \"toTeamId\"]] = df[\n        [\"fromTeamId\", \"toTeamId\"]\n    ].fillna(0)\n\n    df_wide = (\n        df[[\"playerId\", \"date\", \"fromTeamId\", \"toTeamId\", \"typeCode\", \"constant\"]]\n        .pivot_table(\n            index=[\"playerId\", \"date\", \"fromTeamId\", \"toTeamId\"],\n            columns=\"typeCode\",\n            values=\"constant\",\n            aggfunc=\"sum\",\n        )\n        .reset_index()\n    )\n\n    cols_to_fill_na = df.typeCode.unique()\n\n    df_wide[cols_to_fill_na] = df_wide[cols_to_fill_na].fillna(0)\n    df_wide = reduce_mem_usage(df_wide)\n\n    df_wide[cols_to_fill_na] = df_wide[cols_to_fill_na].astype(\"uint8\")\n\n    df_wide = df_wide.sort_values(\n        [\"playerId\", \"date\", \"fromTeamId\", \"toTeamId\"]\n    )\n\n    output = df_wide.groupby(\n        [\"playerId\", \"date\"], as_index=False\n    ).last()\n    \n    output[\"date\"] += pd.Timedelta(days=1)\n    \n    current_cols = output.columns.to_list()\n    holiday_cols_missing = [col for col in TRANSACTIONS_COLS if col not in current_cols]\n    return output.reindex(columns=current_cols + holiday_cols_missing, fill_value=0).drop(columns=[\"date\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.634839Z","iopub.execute_input":"2021-07-27T05:04:54.635180Z","iopub.status.idle":"2021-07-27T05:04:54.650950Z","shell.execute_reply.started":"2021-07-27T05:04:54.635146Z","shell.execute_reply":"2021-07-27T05:04:54.649803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\n    f\"{CUSTOM_DATA_ROOT}/labels.csv\",\n    parse_dates=[\"date\"],\n    dtype={\"target1\": \"float16\",\n            \"target2\": \"float16\",\n            \"target3\": \"float16\",\n            \"target4\": \"float16\"}\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:54.652345Z","iopub.execute_input":"2021-07-27T05:04:54.652717Z","iopub.status.idle":"2021-07-27T05:04:57.570847Z","shell.execute_reply.started":"2021-07-27T05:04:54.652684Z","shell.execute_reply":"2021-07-27T05:04:57.569816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box_scores = pd.read_csv(\n    f\"{CUSTOM_DATA_ROOT}/box_scores.csv\",\n    parse_dates=[\"date\"]\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:57.572299Z","iopub.execute_input":"2021-07-27T05:04:57.572639Z","iopub.status.idle":"2021-07-27T05:04:59.113457Z","shell.execute_reply.started":"2021-07-27T05:04:57.572606Z","shell.execute_reply":"2021-07-27T05:04:59.112373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = reduce_mem_usage(data)\nbox_scores = reduce_mem_usage(box_scores)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:04:59.114830Z","iopub.execute_input":"2021-07-27T05:04:59.115167Z","iopub.status.idle":"2021-07-27T05:05:00.810152Z","shell.execute_reply.started":"2021-07-27T05:04:59.115135Z","shell.execute_reply":"2021-07-27T05:05:00.809086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv(f\"{CUSTOM_DATA_ROOT}/players.csv\",\n                      parse_dates=[\"DOB\", \"mlbDebutDate\"])\n\nplayers_processed = players[PLAYERS_COLS].copy()\n\nplayers_processed[\"birthCountry\"] = pd.Categorical(players_processed[\"birthCountry\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:00.811618Z","iopub.execute_input":"2021-07-27T05:05:00.811922Z","iopub.status.idle":"2021-07-27T05:05:00.834622Z","shell.execute_reply.started":"2021-07-27T05:05:00.811892Z","shell.execute_reply":"2021-07-27T05:05:00.833463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seasons = pd.read_csv(\n    f\"{ROOT}/seasons.csv\",\n    parse_dates=[\n        \"seasonStartDate\",\n        \"seasonEndDate\",\n        \"preSeasonStartDate\",\n        \"preSeasonEndDate\",\n        \"regularSeasonStartDate\",\n        \"regularSeasonEndDate\",\n        \"lastDate1stHalf\",\n        \"allStarDate\",\n        \"firstDate2ndHalf\",\n        \"postSeasonStartDate\",\n        \"postSeasonEndDate\",\n    ],\n)\n\nseasons[\"date\"] = [\n    pd.date_range(start, end)\n    for start, end in zip(seasons.seasonStartDate, seasons.seasonEndDate)\n]\n\nseasons = seasons.explode(\"date\")\n\ndates = pd.DataFrame({\"date\": pd.date_range(seasons.seasonStartDate.min(), seasons.seasonEndDate.max())})\n\nseasons = dates.merge(seasons, on=[\"date\"], how=\"left\")\n\ncols_to_bfill = SEASONS_DAYS_FROM_COLS + SEASONS_DATE_COLS + [\"seasonId\"]\n\nseasons[cols_to_bfill] = seasons[cols_to_bfill].bfill()\n\ndel dates","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:00.835929Z","iopub.execute_input":"2021-07-27T05:05:00.836260Z","iopub.status.idle":"2021-07-27T05:05:00.872780Z","shell.execute_reply.started":"2021-07-27T05:05:00.836229Z","shell.execute_reply":"2021-07-27T05:05:00.871905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"historical_awards = pd.read_csv(f\"{ROOT}/awards.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:00.874086Z","iopub.execute_input":"2021-07-27T05:05:00.874377Z","iopub.status.idle":"2021-07-27T05:05:00.895314Z","shell.execute_reply.started":"2021-07-27T05:05:00.874349Z","shell.execute_reply":"2021-07-27T05:05:00.894387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"teams = pd.read_csv(f\"{ROOT}/teams.csv\")\nteams = reduce_mem_usage(teams)\nteams = teams.rename(columns={\"id\": \"teamId\"})\nteams = teams[TEAMS_COLS]","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:00.896649Z","iopub.execute_input":"2021-07-27T05:05:00.896945Z","iopub.status.idle":"2021-07-27T05:05:00.914991Z","shell.execute_reply.started":"2021-07-27T05:05:00.896918Z","shell.execute_reply":"2021-07-27T05:05:00.914175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_twitter_followers = pd.read_csv(f\"{CUSTOM_DATA_ROOT}/player_twitter_followrs.csv\")\nteam_twitter_followers = pd.read_csv(f\"{CUSTOM_DATA_ROOT}/team_twitter_followrs.csv\")\n\nplayer_twitter_followers = reduce_mem_usage(player_twitter_followers)\nteam_twitter_followers = reduce_mem_usage(team_twitter_followers)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:00.916085Z","iopub.execute_input":"2021-07-27T05:05:00.916501Z","iopub.status.idle":"2021-07-27T05:05:00.955095Z","shell.execute_reply.started":"2021-07-27T05:05:00.916469Z","shell.execute_reply":"2021-07-27T05:05:00.954195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering","metadata":{}},{"cell_type":"code","source":"data = data.sort_values([\"playerId\", \"date\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:00.956115Z","iopub.execute_input":"2021-07-27T05:05:00.956533Z","iopub.status.idle":"2021-07-27T05:05:02.409914Z","shell.execute_reply.started":"2021-07-27T05:05:00.956496Z","shell.execute_reply":"2021-07-27T05:05:02.409110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def engineer_moving_averages(df):\n#     for col in TARGET_COLS:\n#         df[f\"{col}_mean\"] = df.groupby(\"playerId\")[col].transform(\"mean\").astype(\"float16\")\n#         df[f\"{col}_max\"] = df.groupby(\"playerId\")[col].transform(\"max\").astype(\"float16\")\n#         df[f\"{col}_std\"] = df.groupby(\"playerId\")[col].transform(\"std\").astype(\"float16\")\n        \n#     return df\ndef engineer_moving_averages(df):\n    for col in TARGET_COLS:\n        df[f\"{col}_mean\"] = df.groupby(\"playerId\")[col].transform(lambda x: x.shift(1).rolling(7).mean())\n        df[f\"{col}_max\"] = df.groupby(\"playerId\")[col].transform(lambda x: x.shift(1).rolling(7).mean())\n        df[f\"{col}_std\"] = df.groupby(\"playerId\")[col].transform(lambda x: x.shift(1).rolling(7).mean())\n\n    return df\n\n\ndef engineer_seasonality(df):\n    df[\"day_of_week\"] = df[\"date\"].dt.day_of_week.astype(\"uint8\")\n    df[\"week_of_year\"] = df[\"date\"].dt.isocalendar().week.astype(\"uint8\")\n    df[\"month\"] = df[\"date\"].dt.month.astype(\"uint8\")\n    \n    return df\n\n\ndef engineer_player_box_scores(df):\n    df = df.sort_values([\"playerId\", \"date\"], ignore_index=True)\n    for col in BOX_SCORES_COMPARISON_COLS:\n        df[f\"mean_to_this_day_{col}\"] = df.groupby(\"playerId\")[col].transform(lambda x: x.shift(1).expanding(2).mean())\n        df[f\"relative_{col}\"] = df[col] - df[f\"mean_to_this_day_{col}\"]\n        \n        df = df.drop(columns=[f\"mean_to_this_day_{col}\"])\n    \n    cols_to_keep = [col for col in df.columns if re.match(\"date|playerId|relative\", col)]\n    return df[cols_to_keep]\n\n\ndef engineer_seasons(df):\n    df = df.copy()\n    for col in SEASONS_DAYS_FROM_COLS:\n        df[f\"days_from_{col}\"] = (df[\"date\"] - df[col]).dt.days\n\n    for col in SEASONS_DATE_COLS:\n        df[f\"is_{col}\"] = (df[\"date\"] == df[col]).astype(\"uint8\")\n\n    df[\"is_restSeason\"] = np.where(df.days_from_seasonStartDate < 0, 1, 0)\n    cols_to_keep = [\n        \"date\",\n        \"days_from_seasonStartDate\",\n        \"days_from_preSeasonStartDate\",\n        \"days_from_regularSeasonStartDate\",\n        \"days_from_postSeasonStartDate\",\n        \"days_from_postSeasonEndDate\",\n        \"is_allStarDate\",\n        \"is_lastDate1stHalf\",\n        \"is_firstDate2ndHalf\",\n        \"is_seasonEndDate\",\n        \"is_preSeasonEndDate\",\n        \"is_regularSeasonEndDate\",\n        \"is_postSeasonEndDate\",\n        \"is_restSeason\",\n    ]\n\n    return reduce_mem_usage(df[cols_to_keep])\n\n\n\ndef engineer_historical_awards(df):\n    df = df.copy()\n\n    awards_count = df.groupby(\"playerId\")[\"awardName\"].count().rename(\"historicalAwardsCount\")\n    all_star_count = (\n        df.groupby(\"playerId\")[\"awardName\"]\n        .apply(lambda x: (x.str.contains(\"all.*star\", case=False)).sum())\n        .rename(\"historicalAllStarCount\")\n    )\n    mvp_count = (\n        df.groupby(\"playerId\")[\"awardName\"]\n        .apply(lambda x: (x.str.contains(\"mvp|most.*valuable\", case=False)).sum())\n        .rename(\"historicalMvpCount\")\n    )\n    rookie_count = (\n        df.groupby(\"playerId\")[\"awardName\"]\n        .apply(lambda x: (x.str.contains(\"rookie\", case=False)).sum())\n        .rename(\"historicalRookieCount\")\n    )\n\n    return pd.concat(\n        [awards_count, all_star_count, mvp_count, rookie_count], axis=1\n    ).reset_index()\n\n\ndef engineer_awards(df):\n    df = df.copy()\n    \n    df[\"date\"] += pd.Timedelta(days=1)\n    \n    awards_count = df.groupby([\"date\", \"playerId\"])[\"awardName\"].count().rename(\"awardsCount\")\n    all_star_count = (\n        df.groupby([\"date\", \"playerId\"])[\"awardName\"]\n        .apply(lambda x: (x.str.contains(\"all.*star\", case=False)).sum())\n        .rename(\"allStarCount\")\n    )\n    mvp_count = (\n        df.groupby([\"date\", \"playerId\"])[\"awardName\"]\n        .apply(lambda x: (x.str.contains(\"mvp|most.*valuable\", case=False)).sum())\n        .rename(\"mvpCount\")\n    )\n    rookie_count = (\n        df.groupby([\"date\", \"playerId\"])[\"awardName\"]\n        .apply(lambda x: (x.str.contains(\"rookie\", case=False)).sum())\n        .rename(\"rookieCount\")\n    )\n\n    return pd.concat(\n        [awards_count, all_star_count, mvp_count, rookie_count], axis=1\n    ).reset_index()\n\n\ndef engineer_player_twitter(df):\n    df[\"month\"] = df.date.dt.month\n    df[\"year\"] = df.date.dt.year\n    df[\"player_followers_depth\"] = (df[\"numberOfFollowers\"]\n                                    .div(df.groupby(\"month\")[\"numberOfFollowers\"].transform(\"median\"))\n                                   )\n    \n    return df[PLAYER_TWITTER_COLS]\n\n\ndef engineer_team_twitter(df):\n    df[\"month\"] = df.date.dt.month\n    df[\"year\"] = df.date.dt.year\n    df[\"team_followers_depth\"] = (df[\"numberOfFollowers\"]\n                                  .div(df.groupby(\"month\")[\"numberOfFollowers\"].transform(\"median\"))\n                                 )\n    \n    return df[TEAM_TWITTER_COLS]","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:02.411097Z","iopub.execute_input":"2021-07-27T05:05:02.411492Z","iopub.status.idle":"2021-07-27T05:05:02.438612Z","shell.execute_reply.started":"2021-07-27T05:05:02.411462Z","shell.execute_reply":"2021-07-27T05:05:02.437397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1, model2, model3, model4 = [pickle.load(open(f\"../input/modelv8/lightgbm_v8_target{i}.pkl\", \"rb\")) for i in range(1, 5)]","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:02.439859Z","iopub.execute_input":"2021-07-27T05:05:02.440168Z","iopub.status.idle":"2021-07-27T05:05:02.886888Z","shell.execute_reply.started":"2021-07-27T05:05:02.440139Z","shell.execute_reply":"2021-07-27T05:05:02.885765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box_scores = box_scores.drop(columns=[\"gameTimeUTC\", \"jerseyNum\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:02.888431Z","iopub.execute_input":"2021-07-27T05:05:02.888813Z","iopub.status.idle":"2021-07-27T05:05:02.934500Z","shell.execute_reply.started":"2021-07-27T05:05:02.888774Z","shell.execute_reply":"2021-07-27T05:05:02.933678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box_scores_agg = box_scores.groupby([\"date\", \"playerId\"], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:02.935692Z","iopub.execute_input":"2021-07-27T05:05:02.936188Z","iopub.status.idle":"2021-07-27T05:05:03.510805Z","shell.execute_reply.started":"2021-07-27T05:05:02.936156Z","shell.execute_reply":"2021-07-27T05:05:03.510026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box_scores_agg.date += pd.Timedelta(days=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:03.512009Z","iopub.execute_input":"2021-07-27T05:05:03.512353Z","iopub.status.idle":"2021-07-27T05:05:03.520192Z","shell.execute_reply.started":"2021-07-27T05:05:03.512323Z","shell.execute_reply":"2021-07-27T05:05:03.519103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seasons_features = engineer_seasons(seasons)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:03.523491Z","iopub.execute_input":"2021-07-27T05:05:03.523823Z","iopub.status.idle":"2021-07-27T05:05:03.551168Z","shell.execute_reply.started":"2021-07-27T05:05:03.523793Z","shell.execute_reply":"2021-07-27T05:05:03.550212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"historical_awards_features = engineer_historical_awards(historical_awards)\nhistorical_awards_features = reduce_mem_usage(historical_awards_features)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:03.552374Z","iopub.execute_input":"2021-07-27T05:05:03.552669Z","iopub.status.idle":"2021-07-27T05:05:05.656328Z","shell.execute_reply.started":"2021-07-27T05:05:03.552643Z","shell.execute_reply":"2021-07-27T05:05:05.655232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rosters_features = pd.read_csv(f\"{CUSTOM_DATA_ROOT}/rosters.csv\",\n                               parse_dates=[\"date\"])\n\nrosters_features = reduce_mem_usage(rosters_features)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:05.657453Z","iopub.execute_input":"2021-07-27T05:05:05.657740Z","iopub.status.idle":"2021-07-27T05:05:06.652204Z","shell.execute_reply.started":"2021-07-27T05:05:05.657714Z","shell.execute_reply":"2021-07-27T05:05:06.651093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del box_scores\ndel seasons\ndel historical_awards","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:06.653567Z","iopub.execute_input":"2021-07-27T05:05:06.653881Z","iopub.status.idle":"2021-07-27T05:05:06.660602Z","shell.execute_reply.started":"2021-07-27T05:05:06.653850Z","shell.execute_reply":"2021-07-27T05:05:06.659340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"env = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in iter_test:\n    current_date = pd.to_datetime(str(sample_prediction_df.reset_index().date.iat[0]))\n    test_date = current_date + pd.Timedelta(days=1)\n    \n    predictions = sample_prediction_df.reset_index()[\"date_playerId\"].str.split(\"_\", expand=True).rename(columns={0: \"date\", 1: \"playerId\"})\n    predictions[\"date\"] = pd.to_datetime(predictions[\"date\"].astype(str))\n    predictions[\"playerId\"] = predictions[\"playerId\"].astype(\"int32\")\n    \n    test_box_scores = unpack_col_to_df(test_df.playerBoxScores)\n    if not test_box_scores.empty:\n        test_box_scores_agg = process_box_scores(test_box_scores)\n        test_box_scores_agg[\"date\"] += pd.Timedelta(days=1)\n    \n    test_rosters = unpack_col_to_df(test_df.rosters)\n    if not test_rosters.empty:\n        test_rosters = test_rosters.rename(columns={\"gameDate\": \"date\"})\n        test_rosters[\"date\"] += pd.Timedelta(days=1)\n        test_rosters = test_rosters[ROSTERS_COLS]\n    else:\n        test_rosters = pd.DataFrame({'playerId': predictions['playerId']})\n        for col in ROSTERS_COLS:\n            if col == 'playerId': continue\n            elif col == 'date': \n                test_rosters['date'] = test_date\n            else:\n                test_rosters[col] = np.nan\n    \n    awards = unpack_col_to_df(test_df.awards)\n    if not awards.empty:\n        awards = awards.rename(columns={\"awardDate\": \"date\"})\n        awards_features = engineer_awards(awards)\n        awards_features = reduce_mem_usage(awards_features)\n    else:\n        awards_features = pd.DataFrame(columns=AWARDS_COLS, dtype=\"float16\")\n                \n    test_player_twitter_followers = unpack_col_to_df(test_df.playerTwitterFollowers)\n    if not test_player_twitter_followers.empty:\n        test_player_twitter_followers = engineer_player_twitter(test_player_twitter_followers)\n        player_twitter_followers = pd.concat([player_twitter_followers, test_player_twitter_followers])\n\n    test_team_twitter_followers = unpack_col_to_df(test_df.teamTwitterFollowers)\n    if not test_team_twitter_followers.empty:\n        test_team_twitter_followers = engineer_player_twitter(test_team_twitter_followers)\n        team_twitter_followers = pd.concat([team_twitter_followers, test_team_twitter_followers])\n        \n    year_player = player_twitter_followers.year.max()\n    month_player = player_twitter_followers[player_twitter_followers.year.eq(year_player)].month.max()\n    year_team = team_twitter_followers.year.max()\n    month_team = team_twitter_followers[team_twitter_followers.year.eq(year_team)].month.max()\n    \n    player_date_mask = player_twitter_followers.year.eq(year_player) & player_twitter_followers.month.eq(month_player)\n    team_date_mask = team_twitter_followers.year.eq(year_team) & team_twitter_followers.month.eq(month_team)\n    test_player_twitter_followers = player_twitter_followers[player_date_mask][[\"playerId\", \"player_followers_depth\"]]\n    test_team_twitter_followers = team_twitter_followers[team_date_mask][[\"teamId\", \"team_followers_depth\"]]  \n   \n    rosters_features = rosters_features[rosters_features.date <= current_date]\n    rosters_features = pd.concat([rosters_features, test_rosters]).reset_index(drop=True)\n    rosters_features[[\"statusCode\", \"teamId\"]] = rosters_features.groupby(\"playerId\")[[\"statusCode\", \"teamId\"]].ffill()\n    test_rosters_features = rosters_features[rosters_features.date.eq(test_date)].drop(columns=[\"date\"]).copy()\n    \n    test_transactions = unpack_col_to_df(test_df.transactions)\n    if test_transactions.empty:\n        transactions_features = pd.DataFrame(columns=[\"playerId\"] + TRANSACTIONS_COLS)\n    else:\n        transactions_features = process_transactions(test_transactions)\n        \n    box_scores_agg = box_scores_agg[box_scores_agg.date <= current_date]\n    box_scores_agg = pd.concat([box_scores_agg, test_box_scores_agg]).reset_index(drop=True)\n    box_scores_features = pd.concat([engineer_player_box_scores(box_scores_agg), \n                                     box_scores_agg[BOX_SCORES_NUMERIC_COLS],\n                                     box_scores_agg[BOX_SCORES_CATEGORICAL_COLS]], axis=1)\n    \n    \n    data = data[data.date <= current_date]\n    data[\"year\"] = data.date.dt.year.astype(\"uint16\")\n    data = pd.concat([data, predictions]).reset_index(drop=True)\n    data = data.pipe(engineer_moving_averages).pipe(engineer_seasonality)\n        \n    test_data = data.loc[data.date.eq(test_date)].copy()\n    test_data = predictions.merge(\n        test_data.merge(box_scores_features, on=[\"date\", \"playerId\"], how=\"left\"),\n        on=[\"date\", \"playerId\"],\n        how=\"left\",\n        validate=\"1:1\"\n    )\n    test_data = process_players(test_data.merge(players_processed, on=[\"playerId\"], how=\"left\", validate=\"1:1\"))\n    test_data = test_data.merge(seasons_features, on=[\"date\"], how=\"left\", validate=\"m:1\")\n    test_data = test_data.merge(historical_awards_features, on=[\"playerId\"], how=\"left\", validate=\"1:1\")\n    test_data = test_data.merge(awards_features, on=[\"playerId\"], how=\"left\", validate=\"1:1\")\n    test_data = test_data.merge(transactions_features, on=[\"playerId\"], how=\"left\", validate=\"1:1\")\n    test_data = test_data.merge(test_rosters_features, on=[\"playerId\"], how=\"left\", validate=\"1:1\")\n    test_data = test_data.merge(test_player_twitter_followers, on=[\"playerId\"], how=\"left\", validate=\"m:1\")\n    test_data = test_data.merge(test_team_twitter_followers, on=[\"teamId\"], how=\"left\", validate=\"m:1\")\n    test_data = test_data.merge(teams, on=[\"teamId\"], how=\"left\", validate=\"m:1\")\n    \n    test_data[TRANSACTIONS_COLS] = test_data[TRANSACTIONS_COLS].fillna(0)\n    test_data[\"statusCode\"] = pd.Categorical(test_data[\"statusCode\"])\n    test_data[\"leagueId\"] = pd.Categorical(test_data[\"leagueId\"])\n    test_data[\"divisionId\"] = pd.Categorical(test_data[\"divisionId\"])\n    \n    test_data = test_data.drop(columns=[\"year\"])\n    \n    X_test = test_data[FEATURES_COLS]\n    \n    target1 = model1.predict(X_test)\n    target2 = model2.predict(X_test)\n    target3 = model3.predict(X_test)\n    target4 = model4.predict(X_test)\n    \n    sample_prediction_df['target1'] = np.clip(target1, 0, 100)\n    sample_prediction_df['target2'] = np.clip(target2, 0, 100)\n    sample_prediction_df['target3'] = np.clip(target3, 0, 100)\n    sample_prediction_df['target4'] = np.clip(target4, 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0)\n    \n    data.loc[data.date.eq(test_date),'target1'] = np.clip(target1, 0, 100)\n    data.loc[data.date.eq(test_date),'target2'] = np.clip(target2, 0, 100)\n    data.loc[data.date.eq(test_date),'target3'] = np.clip(target3, 0, 100)\n    data.loc[data.date.eq(test_date),'target4'] = np.clip(target4, 0, 100)\n    data = data.fillna(0)\n    \n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-27T05:05:06.662630Z","iopub.execute_input":"2021-07-27T05:05:06.663090Z","iopub.status.idle":"2021-07-27T05:08:14.061190Z","shell.execute_reply.started":"2021-07-27T05:05:06.663025Z","shell.execute_reply":"2021-07-27T05:08:14.060034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}