{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.metrics import mean_absolute_error\nfrom datetime import timedelta\nfrom functools import reduce\nfrom tqdm import tqdm\nimport lightgbm as lgbm\nimport mlb\nimport matplotlib.pyplot as plt\n\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:31:58.273649Z","iopub.execute_input":"2021-10-15T07:31:58.273952Z","iopub.status.idle":"2021-10-15T07:31:58.281653Z","shell.execute_reply.started":"2021-10-15T07:31:58.273921Z","shell.execute_reply":"2021-10-15T07:31:58.280858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"features that we have now","metadata":{}},{"cell_type":"code","source":"BASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\nTRAIN_DIR = Path('../input/mlb-pdef-train-dataset')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:31:58.283091Z","iopub.execute_input":"2021-10-15T07:31:58.283442Z","iopub.status.idle":"2021-10-15T07:31:58.291551Z","shell.execute_reply.started":"2021-10-15T07:31:58.283414Z","shell.execute_reply":"2021-10-15T07:31:58.290335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In the current data, we use only \n> rosters_train.pkl <br>\n> nextDayPlayerEngagement_train.pkl <br>\n> playerBoxScores_train.pkl <br>\n> player_target_stats.csv","metadata":{}},{"cell_type":"code","source":"players = pd.read_csv(BASE_DIR / 'players.csv')\n\nrosters = pd.read_pickle(TRAIN_DIR / 'rosters_train.pkl')\ntargets = pd.read_pickle(TRAIN_DIR / 'nextDayPlayerEngagement_train.pkl')\nfollowers = pd.read_pickle(TRAIN_DIR / 'playerTwitterFollowers_train.pkl')\nteam_followers = pd.read_pickle(TRAIN_DIR / 'teamTwitterFollowers_train.pkl')\nteam_followers = team_followers.rename(columns={'numberOfFollowers': 'teamFollowers'})\nscores = pd.read_pickle(TRAIN_DIR / 'playerBoxScores_train.pkl')\nscores = scores.groupby(['playerId', 'date']).sum().reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:31:58.292647Z","iopub.execute_input":"2021-10-15T07:31:58.293053Z","iopub.status.idle":"2021-10-15T07:32:00.195469Z","shell.execute_reply.started":"2021-10-15T07:31:58.293025Z","shell.execute_reply":"2021-10-15T07:32:00.194414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2   \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64) \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)   \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:32:00.197070Z","iopub.execute_input":"2021-10-15T07:32:00.197369Z","iopub.status.idle":"2021-10-15T07:32:00.211701Z","shell.execute_reply.started":"2021-10-15T07:32:00.197339Z","shell.execute_reply":"2021-10-15T07:32:00.210543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = reduce_mem_usage(players)\nrosters = reduce_mem_usage(rosters)\nfollowers = reduce_mem_usage(followers)\nteam_followers = reduce_mem_usage(team_followers)\nscores = reduce_mem_usage(scores)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:32:00.213197Z","iopub.execute_input":"2021-10-15T07:32:00.213501Z","iopub.status.idle":"2021-10-15T07:32:01.670405Z","shell.execute_reply.started":"2021-10-15T07:32:00.213473Z","shell.execute_reply":"2021-10-15T07:32:01.669400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Rosters EDA","metadata":{}},{"cell_type":"markdown","source":"- playerId - Unique identifier for a player.\n- gameDate - dat of the game\n- teamId - teamId that player is on that date.\n- statusCode - Roster status abbreviation.\n- status - Descriptive roster status.","metadata":{}},{"cell_type":"markdown","source":"# Target EDA","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set_style('whitegrid')\nsns.set(font_scale = 1.5)\nfig, axs = plt.subplots(2,2, figsize = (20, 10))\nsns.kdeplot(ax=axs[0,0], data=targets['target1'])\nsns.kdeplot(ax=axs[0,1], data=targets['target2'])\nsns.kdeplot(ax=axs[1,0], data=targets['target3'])\nsns.kdeplot(ax=axs[1,1], data=targets['target4'])\nbbox = axs[0,0].get_position()\nbbox2 = axs[0,1].get_position()\n\ncenter=(bbox2.x1) * 0.4 + (bbox.x1) * 0.25\nplt.suptitle('Distribution of targets', x = center)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:32:28.063671Z","iopub.execute_input":"2021-10-15T07:32:28.064061Z","iopub.status.idle":"2021-10-15T07:33:14.218298Z","shell.execute_reply.started":"2021-10-15T07:32:28.064029Z","shell.execute_reply":"2021-10-15T07:33:14.214053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw_kde_plot(col = 'target1'):\n    sns.set_style('whitegrid')\n    sns.set(font_scale = 1.5)\n    fig, axs = plt.subplots(2,2, figsize = (15, 10))\n    g = sns.kdeplot(ax=axs[0,0], data=targets[col])\n    g.set_xlabel('original')\n    g = sns.kdeplot(ax=axs[0,1], data=targets[col]**2)\n    g.set_xlabel('squared')\n    g = sns.kdeplot(ax=axs[1,0], data=targets[col]**4)\n    g.set_xlabel('power 4')\n    g = sns.kdeplot(ax=axs[1,1], data = np.log(targets[col]+1))\n    g.set_xlabel('log')\n\n\n\n    bbox = axs[0,0].get_position()\n    bbox2 = axs[0,1].get_position()\n    center=(bbox2.x1) * 0.4 + (bbox.x1) * 0.25\n    plt.suptitle(f'Transformation of {col}', x = center)\n    plt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:33:14.221680Z","iopub.execute_input":"2021-10-15T07:33:14.222008Z","iopub.status.idle":"2021-10-15T07:33:14.232574Z","shell.execute_reply.started":"2021-10-15T07:33:14.221978Z","shell.execute_reply":"2021-10-15T07:33:14.231580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# **Target2 has highest Skewness**\n","metadata":{}},{"cell_type":"code","source":"for col in ['target1', 'target2', 'target3', 'target4']:\n    draw_kde_plot(col)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:33:14.234415Z","iopub.execute_input":"2021-10-15T07:33:14.234764Z","iopub.status.idle":"2021-10-15T07:36:22.758217Z","shell.execute_reply.started":"2021-10-15T07:33:14.234733Z","shell.execute_reply":"2021-10-15T07:36:22.756812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style('whitegrid')\nsns.set(font_scale = 1.5)\n\n\nfig, axs = plt.subplots(1,1, figsize = (20,8))\nsns.lineplot(ax=axs, x = np.arange(1,10001),\n             y = targets.sample(10000, random_state=500)['target1'],\n             legend='full', label = 'target1')\nsns.lineplot(ax=axs, x = np.arange(1,10001),\n             y = targets.sample(10000, random_state=500)['target2'],\n             legend='full', label = 'target2')\nsns.lineplot(ax=axs, x = np.arange(1,10001), \n             y = targets.sample(10000, random_state=500)['target3'], \n             legend='full', label = 'target3')\nsns.lineplot(ax=axs,x = np.arange(1,10001), \n             y = targets.sample(10000, random_state=500)['target4'], \n             legend='full', label = 'target4')\n\nbbox = axs.get_position()\ncenter=0.5*(bbox.x1)\nplt.suptitle('Comparision of targets', x = center)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:22.760200Z","iopub.execute_input":"2021-10-15T07:36:22.760644Z","iopub.status.idle":"2021-10-15T07:36:25.725043Z","shell.execute_reply.started":"2021-10-15T07:36:22.760614Z","shell.execute_reply":"2021-10-15T07:36:25.723924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style('ticks')\nsns.set(font_scale = 1.5)\n\n\nfig, axs = plt.subplots(2,2, figsize = (20,8))\nsns.lineplot(ax=axs[0,0], x = np.arange(1,10001),\n             y = targets.sample(10000, random_state=500)['target1'],\n             legend='full', label = 'target1')\nsns.lineplot(ax=axs[0,1], x = np.arange(1,10001),\n             y = targets.sample(10000, random_state=500)['target2'],\n             legend='full', label = 'target2')\nsns.lineplot(ax=axs[1,0], x = np.arange(1,10001), \n             y = targets.sample(10000, random_state=500)['target3'], \n             legend='full', label = 'target3')\nsns.lineplot(ax=axs[1,1], x = np.arange(1,10001), \n             y = targets.sample(10000, random_state=500)['target4'], \n             legend='full', label = 'target4')\n\nplt.title('Comparision of targets, side by side view')\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:25.726705Z","iopub.execute_input":"2021-10-15T07:36:25.727112Z","iopub.status.idle":"2021-10-15T07:36:29.461260Z","shell.execute_reply.started":"2021-10-15T07:36:25.727074Z","shell.execute_reply":"2021-10-15T07:36:29.460303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets['year'] = pd.to_datetime(targets['date'], format = '%Y%m%d').dt.year","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:29.462476Z","iopub.execute_input":"2021-10-15T07:36:29.462765Z","iopub.status.idle":"2021-10-15T07:36:29.731210Z","shell.execute_reply.started":"2021-10-15T07:36:29.462736Z","shell.execute_reply":"2021-10-15T07:36:29.729968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We have less data for year 4, since we need to predict for the future","metadata":{}},{"cell_type":"markdown","source":"May be we should have different validation strategy","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"targets['year'].value_counts().plot(kind = 'bar')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:29.732617Z","iopub.execute_input":"2021-10-15T07:36:29.732943Z","iopub.status.idle":"2021-10-15T07:36:29.893694Z","shell.execute_reply.started":"2021-10-15T07:36:29.732913Z","shell.execute_reply":"2021-10-15T07:36:29.892563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set(style=\"whitegrid\")\nsns.set(font_scale = 2)\nsns.color_palette(\"Set2\")\n\n\n\nfig, axs = plt.subplots(4,1, figsize = (20,20))\nsns.lineplot(ax=axs[0], \n             x = np.arange(1,10001),\n             data = targets.sample(10000, random_state=100),\n             y = 'target1',\n             hue = 'year',\n             palette='tab10',\n             linewidth=2.5)\n\nsns.lineplot(ax=axs[1], \n             x = np.arange(1,10001),\n             data = targets.sample(10000, random_state=100),\n             y = 'target2',\n             hue = 'year',\n             palette='tab10',\n             linewidth=2.5)\n\nsns.lineplot(ax=axs[2], \n             x = np.arange(1,10001),\n             data = targets.sample(10000, random_state=100),\n             y = 'target3',\n             hue = 'year',\n             palette='tab10',\n             linewidth=2.5)\n\nsns.lineplot(ax=axs[3], \n             x = np.arange(1,10001),\n             data = targets.sample(10000, random_state=100),\n             y = 'target4',\n             hue = 'year',\n             palette='tab10',\n             linewidth=2.5)\n\nbbox = axs[0].get_position()\ncenter=0.5*(bbox.x1)\nplt.suptitle('targets over years', x = center)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:33.703751Z","iopub.execute_input":"2021-10-15T07:36:33.704052Z","iopub.status.idle":"2021-10-15T07:36:37.417914Z","shell.execute_reply.started":"2021-10-15T07:36:33.704025Z","shell.execute_reply":"2021-10-15T07:36:37.416917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef plot_target_for_player(col = 'target1', playerid = 683734):\n    \n    sns.set(style=\"whitegrid\")\n    sns.set(font_scale = 2)\n    sns.color_palette(\"Set2\")\n\n    fig, axs = plt.subplots(1,1, figsize = (20,8))\n\n    sns.lineplot(ax=axs, x = np.arange(365),\n                 data = targets[((targets.year==2018) & (targets.playerId==playerid))],\n                 y = col,\n                 label = '2018',\n                 linewidth=2.5)\n\n    sns.lineplot(ax=axs, \n                 x =  np.arange(365),\n                 data = targets[((targets.year==2019) & (targets.playerId==playerid))],\n                 y = col,\n                 label = '2019',\n                 linewidth=2.5)\n\n    sns.lineplot(ax=axs, \n                 x =  np.arange(366),\n                 data = targets[((targets.year==2020) & (targets.playerId==playerid))],\n                 y = col,\n                 label = '2020',\n                 linewidth=2.5)\n\n    sns.lineplot(ax=axs, \n                 x =  np.arange(120),\n                 data = targets[((targets.year==2021) & (targets.playerId==playerid))],\n                 y = col,\n                 label = '2021',\n                 linewidth=2.5)\n    \n    bbox = axs.get_position()\n    center=0.5*(bbox.x1)\n    plt.suptitle(f'player Id {playerid}', x = center)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:37.419089Z","iopub.execute_input":"2021-10-15T07:36:37.419353Z","iopub.status.idle":"2021-10-15T07:36:37.430900Z","shell.execute_reply.started":"2021-10-15T07:36:37.419327Z","shell.execute_reply":"2021-10-15T07:36:37.429594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There is definitely seasonality for targets, seems like we can remove year 2018 from modelling","metadata":{}},{"cell_type":"code","source":"plot_target_for_player()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:37.432037Z","iopub.execute_input":"2021-10-15T07:36:37.432311Z","iopub.status.idle":"2021-10-15T07:36:37.879112Z","shell.execute_reply.started":"2021-10-15T07:36:37.432284Z","shell.execute_reply":"2021-10-15T07:36:37.878221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target2')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:37.880265Z","iopub.execute_input":"2021-10-15T07:36:37.880517Z","iopub.status.idle":"2021-10-15T07:36:38.318442Z","shell.execute_reply.started":"2021-10-15T07:36:37.880493Z","shell.execute_reply":"2021-10-15T07:36:38.317701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target3')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:38.319415Z","iopub.execute_input":"2021-10-15T07:36:38.319794Z","iopub.status.idle":"2021-10-15T07:36:38.755701Z","shell.execute_reply.started":"2021-10-15T07:36:38.319767Z","shell.execute_reply":"2021-10-15T07:36:38.755070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target4')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:38.756646Z","iopub.execute_input":"2021-10-15T07:36:38.757017Z","iopub.status.idle":"2021-10-15T07:36:39.222991Z","shell.execute_reply.started":"2021-10-15T07:36:38.756989Z","shell.execute_reply":"2021-10-15T07:36:39.221941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target1',477132)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:39.224221Z","iopub.execute_input":"2021-10-15T07:36:39.224542Z","iopub.status.idle":"2021-10-15T07:36:39.689337Z","shell.execute_reply.started":"2021-10-15T07:36:39.224505Z","shell.execute_reply":"2021-10-15T07:36:39.688365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target2',477132)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:39.690540Z","iopub.execute_input":"2021-10-15T07:36:39.690834Z","iopub.status.idle":"2021-10-15T07:36:40.187517Z","shell.execute_reply.started":"2021-10-15T07:36:39.690794Z","shell.execute_reply":"2021-10-15T07:36:40.186503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target3',477132)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:40.188922Z","iopub.execute_input":"2021-10-15T07:36:40.189197Z","iopub.status.idle":"2021-10-15T07:36:40.676714Z","shell.execute_reply.started":"2021-10-15T07:36:40.189170Z","shell.execute_reply":"2021-10-15T07:36:40.675737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_target_for_player('target4',477132)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:40.677927Z","iopub.execute_input":"2021-10-15T07:36:40.678221Z","iopub.status.idle":"2021-10-15T07:36:41.160459Z","shell.execute_reply.started":"2021-10-15T07:36:40.678191Z","shell.execute_reply":"2021-10-15T07:36:41.159702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There seems to be relationship between number of awards and targets, higher awards the player is popular","metadata":{}},{"cell_type":"code","source":"from scipy.stats import boxcox\nxt, _ = boxcox(targets['target1'].values + 1)\nsns.distplot(xt)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:41.161640Z","iopub.execute_input":"2021-10-15T07:36:41.162027Z","iopub.status.idle":"2021-10-15T07:36:57.576320Z","shell.execute_reply.started":"2021-10-15T07:36:41.161995Z","shell.execute_reply":"2021-10-15T07:36:57.575622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xt, _ = boxcox(targets['target4'].values + 1)\nsns.distplot(xt)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:36:57.577390Z","iopub.execute_input":"2021-10-15T07:36:57.577840Z","iopub.status.idle":"2021-10-15T07:37:14.524134Z","shell.execute_reply.started":"2021-10-15T07:36:57.577795Z","shell.execute_reply":"2021-10-15T07:37:14.523436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xt, _ = boxcox(targets['target3'].values + 1)\nsns.distplot(xt)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:37:14.525259Z","iopub.execute_input":"2021-10-15T07:37:14.525698Z","iopub.status.idle":"2021-10-15T07:37:30.394920Z","shell.execute_reply.started":"2021-10-15T07:37:14.525668Z","shell.execute_reply":"2021-10-15T07:37:30.393918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**Target4 has highest correlation with Twitter follower count**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:37:30.396297Z","iopub.execute_input":"2021-10-15T07:37:30.396594Z","iopub.status.idle":"2021-10-15T07:37:30.400421Z","shell.execute_reply.started":"2021-10-15T07:37:30.396564Z","shell.execute_reply":"2021-10-15T07:37:30.399550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"followers_agg =followers.groupby('playerId')['numberOfFollowers'].agg('median').reset_index()\ntargets_agg =targets.groupby('playerId')[['target1', 'target2', 'target3', 'target4']].agg('median').reset_index()\nfollowers_agg.columns = ['playerId', '#Followers']\nfollowers_agg = pd.merge(followers_agg, targets_agg, on = ['playerId'], how = 'left')\nplt.figure(figsize=(10, 2))\nplt.xticks(rotation=45)\nplt.suptitle(\"Median Target vs Median Twitter Followers\", fontsize =15)\n\ncorr = followers_agg.drop(columns =['playerId']).corr()\n#mask = np.triu(np.ones_like(corr, dtype=bool))\n\nx_axis_labels = ['#Followers', 'target1','target2', 'target3', 'target4'] \nsns.heatmap(np.array(corr['#Followers']).reshape((1,5)),\n            annot = True,\n            xticklabels=x_axis_labels,\n            vmin=0,\n            vmax=1,\n            center= 0,\n            cmap=\"RdYlGn\"\n       )","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:37:30.403619Z","iopub.execute_input":"2021-10-15T07:37:30.403939Z","iopub.status.idle":"2021-10-15T07:37:31.265556Z","shell.execute_reply.started":"2021-10-15T07:37:30.403910Z","shell.execute_reply":"2021-10-15T07:37:31.264619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"awards = pd.read_csv(TRAIN_DIR / 'awards_train.csv')\n\nawards_agg = awards.groupby('playerId')['awardId'].agg('count').reset_index()\nawards_agg.columns = ['playerId', '#Awards']\ntargets_agg =targets.groupby('playerId')[['target1', 'target2', 'target3', 'target4']].agg('median').reset_index()\nfollowers_agg = pd.merge(awards_agg, targets_agg, on = ['playerId'], how = 'left')\nplt.figure(figsize=(10, 2))\nplt.xticks(rotation=45)\nplt.suptitle(\"Total number of awards vs Median targets\", fontsize =15)\ncorr = followers_agg.drop(columns =['playerId']).corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\n\nx_axis_labels = ['#Awards', 'target1','target2', 'target3', 'target4'] \nsns.heatmap(np.array(corr['#Awards']).reshape((1,5)),\n            annot = True,\n            xticklabels=x_axis_labels,\n            vmin=0,\n            vmax=1,\n            center= 0,\n            cmap=\"RdYlGn\"\n       )\ndel awards","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:37:31.267169Z","iopub.execute_input":"2021-10-15T07:37:31.267439Z","iopub.status.idle":"2021-10-15T07:37:31.898420Z","shell.execute_reply.started":"2021-10-15T07:37:31.267413Z","shell.execute_reply":"2021-10-15T07:37:31.897440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_cols = scores.columns.tolist()\n\nscores_cols = [col for col in scores_cols if col not in ['playerId', 'date', 'home', 'gamePk', 'teamId', 'battingOrder']]\nfor col in scores_cols:\n            scores_agg = scores.groupby('playerId')[col].agg('sum').reset_index()\n            scores_agg.columns = ['playerId', \"#\"+col]\n            targets_agg =targets.groupby('playerId')[['target1', 'target2', 'target3', 'target4']].agg('median').reset_index()\n            scores_agg = pd.merge(scores_agg, targets_agg, on = ['playerId'], how = 'left')\n            plt.figure(figsize=(10, 2))\n            plt.suptitle(f\"Total {col} vs Median targets\", fontsize =15)\n            corr = scores_agg.drop(columns =['playerId']).corr()\n            mask = np.triu(np.ones_like(corr, dtype=bool))\n            sns.set(font_scale=1.4)\n            plt.xticks(rotation=45)\n            \n            x_axis_labels = [\"#\"+col, 'target1','target2', 'target3', 'target4'] \n\n            sns.heatmap(np.array(corr[\"#\"+col]).reshape((1,5)),\n            annot = True,\n            xticklabels=x_axis_labels,\n            vmin=0,\n            vmax=1,\n            center= 0,\n            cmap=\"RdYlGn\")           \n            \n            plt.show()\n            plt.close()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:37:31.899586Z","iopub.execute_input":"2021-10-15T07:37:31.899862Z","iopub.status.idle":"2021-10-15T07:38:14.752785Z","shell.execute_reply.started":"2021-10-15T07:37:31.899835Z","shell.execute_reply":"2021-10-15T07:38:14.751704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**about players followers**: Followers is highly correlated with Target4. <br>\n**about player box scores**: Most of box scores are highly correlated with Target2. <br>\n**about awards**: Awards are highly correlated with target1","metadata":{}},{"cell_type":"code","source":"seasons_df = pd.read_csv(BASE_DIR / 'seasons.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:30:38.760382Z","iopub.execute_input":"2021-10-15T07:30:38.760675Z","iopub.status.idle":"2021-10-15T07:30:38.776112Z","shell.execute_reply.started":"2021-10-15T07:30:38.760647Z","shell.execute_reply":"2021-10-15T07:30:38.775126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets['year'] = pd.to_datetime(targets['date'], format = '%Y%m%d').dt.year\ntargets = pd.merge(targets,\n                   seasons_df,\n                   how = 'left',\n                   left_on = 'year',\n                   right_on = 'seasonId')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:38:14.754052Z","iopub.execute_input":"2021-10-15T07:38:14.754338Z","iopub.status.idle":"2021-10-15T07:38:15.659393Z","shell.execute_reply.started":"2021-10-15T07:38:14.754312Z","shell.execute_reply":"2021-10-15T07:38:15.658274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets['engagementMetricsDate'] = pd.to_datetime(targets['engagementMetricsDate'], format='%Y-%m-%d').dt.date\ntargets['seasonEndDate'] = pd.to_datetime(targets['seasonEndDate'], format='%Y-%m-%d').dt.date\ntargets['seasonStartDate'] = pd.to_datetime(targets['seasonStartDate'], format='%Y-%m-%d').dt.date\ntargets['preSeasonEndDate'] = pd.to_datetime(targets['preSeasonEndDate'], format='%Y-%m-%d').dt.date\ntargets['preSeasonStartDate'] = pd.to_datetime(targets['preSeasonStartDate'], format='%Y-%m-%d').dt.date\ntargets['regularSeasonStartDate'] = pd.to_datetime(targets['regularSeasonStartDate'], format='%Y-%m-%d').dt.date\ntargets['regularSeasonEndDate'] = pd.to_datetime(targets['regularSeasonEndDate'], format='%Y-%m-%d').dt.date\ntargets['days_to_season_end'] = (targets.seasonEndDate - targets.engagementMetricsDate).dt.days\ntargets['days_to_season_start'] = (targets.seasonStartDate - targets.engagementMetricsDate).dt.days\n\ntargets['during_season'] = np.where(((targets.seasonStartDate <= targets.engagementMetricsDate)\n                                   & (targets.seasonEndDate  >= targets.engagementMetricsDate)), 1, 0)\ntargets['during_preseason'] = np.where(((targets.preSeasonStartDate <= targets.engagementMetricsDate)\n                                   & (targets.preSeasonEndDate  >= targets.engagementMetricsDate)), 1, 0)\n\ntargets['during_regseason'] = np.where(((targets.regularSeasonStartDate <= targets.engagementMetricsDate)\n                                   & (targets.regularSeasonEndDate  >= targets.engagementMetricsDate)), 1, 0)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:38:15.660780Z","iopub.execute_input":"2021-10-15T07:38:15.661057Z","iopub.status.idle":"2021-10-15T07:38:50.868580Z","shell.execute_reply.started":"2021-10-15T07:38:15.661031Z","shell.execute_reply":"2021-10-15T07:38:50.867568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\nsns.set(font_scale = 1)\nsns.color_palette(\"Set2\")\n\nfor target in ['target1', 'target2', 'target3', 'target4']:\n    df = targets[['days_to_season_start', target, 'year']].groupby(['days_to_season_start', 'year'])[target].agg('median').reset_index()\n    for year in [2018, 2019, 2020, 2021]:\n        plt.figure(figsize=(20, 5))\n        plot_ = sns.barplot(data = df[df.year==year], x = 'days_to_season_start', y = target)\n        for label in plot_.get_xticklabels():\n            if np.int(label.get_text()) % 10 == 0:  \n                label.set_visible(True)\n            else:\n                label.set_visible(False)\n        plt.title(f'{target} - year {year}', fontsize = 20)\n        plt.xlabel(\"days to season start\", fontsize = 15)\n        plt.ylabel(f\"{target}\", fontsize = 15)\n\n        plt.show()\n        plt.close()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T07:38:50.870091Z","iopub.execute_input":"2021-10-15T07:38:50.870364Z","iopub.status.idle":"2021-10-15T07:39:34.244300Z","shell.execute_reply.started":"2021-10-15T07:38:50.870335Z","shell.execute_reply":"2021-10-15T07:39:34.243166Z"},"trusted":true},"execution_count":null,"outputs":[]}]}