{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom datetime import datetime, timedelta\n\ntrain_df = pd.read_csv(\"/kaggle/input/mlb-player-digital-engagement-forecasting/train.csv\", nrows=500)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:26.745228Z","iopub.execute_input":"2021-07-02T03:59:26.745740Z","iopub.status.idle":"2021-07-02T03:59:54.147036Z","shell.execute_reply.started":"2021-07-02T03:59:26.745635Z","shell.execute_reply":"2021-07-02T03:59:54.146148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players_df = pd.read_csv(\"/kaggle/input/mlb-player-digital-engagement-forecasting/players.csv\")\nplayers_df[\"position_and_birth_country\"] = players_df.apply(\n    lambda x: \"{0}_{1}\".format(x[\"birthCountry\"], x[\"primaryPositionName\"]), axis=1)\nplayers_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:54.148783Z","iopub.execute_input":"2021-07-02T03:59:54.149270Z","iopub.status.idle":"2021-07-02T03:59:54.221814Z","shell.execute_reply.started":"2021-07-02T03:59:54.149221Z","shell.execute_reply":"2021-07-02T03:59:54.220763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\n\ndef calc_debut_age(r):\n    try:\n        value = int((datetime.strptime(str(r[\"mlbDebutDate\"]), \"%Y-%m-%d\") - datetime.strptime(str(r[\"DOB\"]), \"%Y-%m-%d\")).days)\n    except Exception:\n        value = None\n        pass\n    return value\n\nplayers_df[\"debut_age_days\"] = players_df.apply(lambda x: calc_debut_age(x),axis=1)\n\ndebut_age_dic = dict(zip(players_df[\"playerId\"], players_df[\"debut_age_days\"]))\nprint(\"Created debut_age_dic\", len(debut_age_dic))","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:54.223302Z","iopub.execute_input":"2021-07-02T03:59:54.223587Z","iopub.status.idle":"2021-07-02T03:59:54.325942Z","shell.execute_reply.started":"2021-07-02T03:59:54.223560Z","shell.execute_reply":"2021-07-02T03:59:54.325153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_characteristics_category_dic = dict()\n\ncategory_headers = [\"birthCountry\", \"primaryPositionName\", \"position_and_birth_country\"]\n\nfor category_header in category_headers:\n    player_characteristics_category_dic[category_header] = dict(zip(players_df[\"playerId\"],\n                                                                    players_df[category_header]))\n    print(category_header, len(player_characteristics_category_dic[category_header]))","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:54.326991Z","iopub.execute_input":"2021-07-02T03:59:54.327429Z","iopub.status.idle":"2021-07-02T03:59:54.336923Z","shell.execute_reply.started":"2021-07-02T03:59:54.327396Z","shell.execute_reply":"2021-07-02T03:59:54.335698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime, timedelta\n\nplayer_birthday_dic = dict(zip(players_df[\"playerId\"], players_df[\"DOB\"]))\nplayer_debut_dic = dict(zip(players_df[\"playerId\"], players_df[\"mlbDebutDate\"]))\n\nprint(\"player_debut_dic\", len(player_debut_dic))\nprint(\"player_birthday_dic\", len(player_birthday_dic))","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:54.338249Z","iopub.execute_input":"2021-07-02T03:59:54.338535Z","iopub.status.idle":"2021-07-02T03:59:54.360422Z","shell.execute_reply.started":"2021-07-02T03:59:54.338507Z","shell.execute_reply":"2021-07-02T03:59:54.359170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_age_dic(player_age_dic, key, reference_dic, min_time, max_time):\n    \n    print(\"Building age dic...\", key)\n\n    continue_loop = True\n    j = 0\n    data_added = 0\n\n    while continue_loop:\n\n        target_date = datetime.strptime(str(min_time), \"%Y%m%d\") + timedelta(days=j)\n        target_date_key = datetime.strftime(target_date, \"%Y%m%d\")\n\n        for player in reference_dic:\n\n            try:\n\n                baseline_day = datetime.strptime(str(reference_dic[player]), \"%Y-%m-%d\")\n                days_since_baseline_day = (target_date - baseline_day).days\n                \n                if days_since_baseline_day > 0:\n                    player_age_dic[(target_date_key, int(player), key)] = days_since_baseline_day\n                    data_added += 1\n                else:\n                    player_age_dic[(target_date_key, int(player), key)] = 0\n\n            except Exception as e:\n                #print(e)\n                player_age_dic[(target_date_key, int(player), key)] = None\n                pass\n\n        j += 1\n\n        if target_date_key == max_time:\n            continue_loop = False\n\n    print(\"Success, data_added\", key, data_added)\n    \n    return player_age_dic","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:54.362194Z","iopub.execute_input":"2021-07-02T03:59:54.362621Z","iopub.status.idle":"2021-07-02T03:59:54.377376Z","shell.execute_reply.started":"2021-07-02T03:59:54.362577Z","shell.execute_reply":"2021-07-02T03:59:54.376188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_time = \"20170101\"\nmax_time = \"20220101\"\n\nplayer_age_dic = dict()\n\nplayer_age_dic = build_age_dic(player_age_dic,\n                               key=\"days_since_birthday\",\n                               reference_dic=player_birthday_dic,\n                               min_time=min_time,\n                               max_time=max_time)\n\nplayer_age_dic = build_age_dic(player_age_dic,\n                               key=\"days_since_debut\",\n                               reference_dic=player_debut_dic,\n                               min_time=min_time,\n                               max_time=max_time)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T03:59:54.378760Z","iopub.execute_input":"2021-07-02T03:59:54.379263Z","iopub.status.idle":"2021-07-02T04:01:35.816804Z","shell.execute_reply.started":"2021-07-02T03:59:54.379215Z","shell.execute_reply":"2021-07-02T04:01:35.816027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"awards_df = pd.read_csv(\"/kaggle/input/mlb-player-digital-engagement-forecasting/awards.csv\")\nawards_df[\"award_date_dateobj\"] = awards_df[\"awardDate\"].map(lambda x: datetime.strptime(x, \"%Y-%m-%d\"))\nawards_df.sort_values(by=\"awardDate\", ascending=True, inplace=True)\nawards_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:01:35.819033Z","iopub.execute_input":"2021-07-02T04:01:35.819448Z","iopub.status.idle":"2021-07-02T04:01:36.011912Z","shell.execute_reply.started":"2021-07-02T04:01:35.819416Z","shell.execute_reply":"2021-07-02T04:01:36.011164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\n\nplayer_award_dic = defaultdict(list)\n\nfor i, row in awards_df.iterrows():\n    data = [row[\"award_date_dateobj\"], row[\"awardId\"]]\n    player_award_dic[row[\"playerId\"]].append(data)\n\nprint(\"Created player award dic\", len(player_award_dic))\n\nfor p in [q for q in player_award_dic][0:10]:\n    print(p, len(player_award_dic[p]))","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:01:36.013254Z","iopub.execute_input":"2021-07-02T04:01:36.013872Z","iopub.status.idle":"2021-07-02T04:01:37.167903Z","shell.execute_reply.started":"2021-07-02T04:01:36.013803Z","shell.execute_reply":"2021-07-02T04:01:37.166714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper function to unpack json found in daily data\ndef unpack_json(json_str):\n    return pd.DataFrame() if pd.isna(json_str) else pd.read_json(json_str)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:01:37.169707Z","iopub.execute_input":"2021-07-02T04:01:37.170180Z","iopub.status.idle":"2021-07-02T04:01:37.177062Z","shell.execute_reply.started":"2021-07-02T04:01:37.170130Z","shell.execute_reply":"2021-07-02T04:01:37.175470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**FEATURE ENGINEERING...**\n\ndictionaries we can use for feature engineering:\n- player_characteristics_category_dic\n- player_age_dic (days since birth, days since debut)\n- player_award_dic","metadata":{}},{"cell_type":"code","source":"def feature_engineering_prediction_df(input_df, input_sample_prediction_df, row_threshold):\n    \n    rows_processed = 0\n    \n    if input_sample_prediction_df.shape[0] > 0:\n        columns_to_check = [\"date\", \"date_playerId\", \"target1\", \"target2\", \"target2\", \"target3\", \"target4\"]\n        for column_to_check in columns_to_check:\n            assert column_to_check in input_sample_prediction_df.columns.values\n    \n    assert input_df.shape[0] > 0\n    \n    if input_sample_prediction_df.shape[0] == 0:\n        \n        assert \"nextDayPlayerEngagement\" in input_df.columns.values\n        assert \"date\" in input_df.columns.values\n        final_sample_prediction_df = pd.DataFrame()\n        \n        for i, row in input_df.iterrows():\n            \n            next_day_player_engagement_df = unpack_json(row[\"nextDayPlayerEngagement\"])\n            if final_sample_prediction_df.shape[0] == 0 and next_day_player_engagement_df.shape[0] > 0:\n                final_sample_prediction_df = next_day_player_engagement_df\n            else:\n                assert len(final_sample_prediction_df.columns.values) == len(next_day_player_engagement_df.columns.values)\n                intersection_set = set.intersection(set(final_sample_prediction_df.columns.values),\n                                                    set(next_day_player_engagement_df.columns.values))\n                assert len(intersection_set) == len(final_sample_prediction_df.columns.values)\n                assert len(intersection_set) == len(next_day_player_engagement_df.columns.values)\n                final_sample_prediction_df = final_sample_prediction_df.append(next_day_player_engagement_df)\n        \n            rows_processed += 1\n            \n            if row_threshold > 0 and rows_processed >= row_threshold:\n                break\n        \n        assert final_sample_prediction_df.shape[0] > 0\n        \n        final_sample_prediction_df[\"date\"] = final_sample_prediction_df[\"engagementMetricsDate\"].map(\n            lambda x: (datetime.strptime(x, \"%Y-%m-%d\") - timedelta(days=1)).strftime(\"%Y%m%d\"))\n        final_sample_prediction_df[\"date_playerId\"] = final_sample_prediction_df.apply(\n            lambda x: \"{0}_{1}\".format(datetime.strptime(x[\"engagementMetricsDate\"],\n                                                         \"%Y-%m-%d\").strftime(\"%Y%m%d\"),\n                                       x[\"playerId\"]), axis=1)\n        \n        final_headers = [\"date\", \"date_playerId\", \"target1\", \"target2\", \"target3\", \"target4\"]\n        final_sample_prediction_df = final_sample_prediction_df[final_headers].copy()\n        \n    else:\n        \n        final_sample_prediction_df = input_sample_prediction_df.copy()\n    \n    print(\"Feature engineering complete, rows processed = {}\".format(rows_processed))\n    \n    return final_sample_prediction_df","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:01:37.179061Z","iopub.execute_input":"2021-07-02T04:01:37.179569Z","iopub.status.idle":"2021-07-02T04:01:37.197129Z","shell.execute_reply.started":"2021-07-02T04:01:37.179517Z","shell.execute_reply":"2021-07-02T04:01:37.195837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sample_prediction_df = feature_engineering_prediction_df(input_df=train_df,\n                                                               input_sample_prediction_df=pd.DataFrame(),\n                                                               row_threshold=200)\nfinal_sample_prediction_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:01:37.198842Z","iopub.execute_input":"2021-07-02T04:01:37.199518Z","iopub.status.idle":"2021-07-02T04:02:05.431281Z","shell.execute_reply.started":"2021-07-02T04:01:37.199468Z","shell.execute_reply":"2021-07-02T04:02:05.430144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_feature_ages(df, player_age_dic):\n    \n    df[\"feature_days_since_dob\"] = df.apply(\n        lambda x: player_age_dic[(x[\"date\"],\n                                  x[\"playerId\"],\n                                  \"days_since_birthday\")] if (x[\"date\"],\n                                                              x[\"playerId\"],\n                                                              \"days_since_birthday\") in player_age_dic else None,\n        axis=1)\n    \n    df[\"feature_days_since_debut\"] = df.apply(\n        lambda x: player_age_dic[(x[\"date\"],\n                                  x[\"playerId\"],\n                                  \"days_since_debut\")] if (x[\"date\"],\n                                                           x[\"playerId\"],\n                                                           \"days_since_debut\") in player_age_dic else None,\n        axis=1)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:02:05.432890Z","iopub.execute_input":"2021-07-02T04:02:05.433241Z","iopub.status.idle":"2021-07-02T04:02:05.440962Z","shell.execute_reply.started":"2021-07-02T04:02:05.433209Z","shell.execute_reply":"2021-07-02T04:02:05.439934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineering_model_df(input_prediction_df,\n                                 player_age_dic):\n    \n    print(\"input_prediction_df columns\", input_prediction_df.columns.values)\n    \n    assert input_prediction_df.shape[0] > 0\n    \n    if \"date\" not in input_prediction_df.columns.values:\n        assert \"date_playerId\" in input_prediction_df.columns.values\n        input_prediction_df[\"date\"] = input_prediction_df[\"date_playerId\"].map(\n            lambda x: (datetime.strptime(str(x).split(\"_\")[0], \"%Y%m%d\") - timedelta(days=1)).strftime(\"%Y%m%d\"))\n    \n    required_headers = [\"date\", \"date_playerId\", \"target1\", \"target2\", \"target3\", \"target4\"]\n    for required_header in required_headers:\n        assert required_header in input_prediction_df.columns.values\n    \n    final_model_df = input_prediction_df[required_headers].copy()\n    final_model_df[\"playerId\"] = final_model_df[\"date_playerId\"].map(lambda x: int(str(x).split(\"_\")[1]))\n    \n    final_model_df = add_feature_ages(final_model_df, player_age_dic)\n    \n    return final_model_df","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:02:05.442279Z","iopub.execute_input":"2021-07-02T04:02:05.442680Z","iopub.status.idle":"2021-07-02T04:02:05.457409Z","shell.execute_reply.started":"2021-07-02T04:02:05.442647Z","shell.execute_reply":"2021-07-02T04:02:05.456132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model_df = feature_engineering_model_df(input_prediction_df=final_sample_prediction_df,\n                                              player_age_dic=player_age_dic)\n\nfinal_model_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:02:05.458832Z","iopub.execute_input":"2021-07-02T04:02:05.459288Z","iopub.status.idle":"2021-07-02T04:02:30.100628Z","shell.execute_reply.started":"2021-07-02T04:02:05.459243Z","shell.execute_reply":"2021-07-02T04:02:30.099399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# then we train the model...\n\nfrom lightgbm import LGBMRegressor\n\nmodel_dic = dict()\nfeature_header_dic = dict()\n\nfeature_headers = [f for f in final_model_df.columns.values if f.find(\"feature_\") != -1]\nassert len(feature_headers) > 0\n\nfor target_header in [\"target1\", \"target2\", \"target3\", \"target4\"]:\n    \n    print(\"Training model\", target_header)\n    clf = LGBMRegressor()\n    clf.fit(final_model_df[feature_headers], final_model_df[target_header])\n    model_dic[target_header] = clf\n    feature_header_dic[target_header] = feature_headers\n    del clf","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:02:30.102207Z","iopub.execute_input":"2021-07-02T04:02:30.102553Z","iopub.status.idle":"2021-07-02T04:02:34.653406Z","shell.execute_reply.started":"2021-07-02T04:02:30.102521Z","shell.execute_reply":"2021-07-02T04:02:34.652441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in iter_test:\n    \n    final_model_df = feature_engineering_model_df(input_prediction_df=sample_prediction_df,\n                                                  player_age_dic=player_age_dic)\n    \n    for target_header in [\"target1\", \"target2\", \"target3\", \"target4\"]:\n        feature_headers = feature_header_dic[target_header]\n        final_model_df[target_header] = model_dic[target_header].predict(final_model_df[feature_headers])\n        sample_prediction_df[target_header] = final_model_df[target_header].values\n    \n    final_headers = [\"date_playerId\", \"target1\", \"target2\", \"target3\", \"target4\"]\n    sample_prediction_df = sample_prediction_df[final_headers].copy()\n    \n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:02:34.654605Z","iopub.execute_input":"2021-07-02T04:02:34.655101Z","iopub.status.idle":"2021-07-02T04:02:36.961293Z","shell.execute_reply.started":"2021-07-02T04:02:34.655063Z","shell.execute_reply":"2021-07-02T04:02:36.959991Z"},"trusted":true},"execution_count":null,"outputs":[]}]}