{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install seedir","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:15:37.607757Z","iopub.execute_input":"2021-06-27T17:15:37.608418Z","iopub.status.idle":"2021-06-27T17:15:45.664444Z","shell.execute_reply.started":"2021-06-27T17:15:37.608324Z","shell.execute_reply":"2021-06-27T17:15:45.663229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nimport warnings\nimport seedir as sd\nwarnings.filterwarnings(\"ignore\")\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:15:45.666286Z","iopub.execute_input":"2021-06-27T17:15:45.666586Z","iopub.status.idle":"2021-06-27T17:15:46.643297Z","shell.execute_reply.started":"2021-06-27T17:15:45.666552Z","shell.execute_reply":"2021-06-27T17:15:46.642135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/mlb-player-digital-engagement-forecasting'\nsd.seedir(data_dir, style='emoji')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:15:46.644870Z","iopub.execute_input":"2021-06-27T17:15:46.645165Z","iopub.status.idle":"2021-06-27T17:15:46.661660Z","shell.execute_reply.started":"2021-06-27T17:15:46.645137Z","shell.execute_reply":"2021-06-27T17:15:46.660522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Static files that do not change with time:\n* players.csv \n* teams.csv\n* seasons.csv\n* awards.csv\n\nDaily data:\n* train.csv\n\nExample test and submission:\n* example_test.csv\n* example_sample_submission.csv\n\nThe test data arrives in a data frame identical in format to train.csv, except it does not contain the target values. It means that all 4 targets are in column **nextDayPlayerEngagement** in the train.csv and it is represented as a big string\n\n\n","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(f'{data_dir}/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:15:50.653604Z","iopub.execute_input":"2021-06-27T17:15:50.654034Z","iopub.status.idle":"2021-06-27T17:17:06.453232Z","shell.execute_reply.started":"2021-06-27T17:15:50.653997Z","shell.execute_reply":"2021-06-27T17:17:06.452180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['date'] == 20180101]['nextDayPlayerEngagement'][0][:500]","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:17:06.455285Z","iopub.execute_input":"2021-06-27T17:17:06.455737Z","iopub.status.idle":"2021-06-27T17:17:06.483800Z","shell.execute_reply.started":"2021-06-27T17:17:06.455689Z","shell.execute_reply":"2021-06-27T17:17:06.482634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is clear that we need somehow to preprocess these strings into data frames, e.g. create the unnested data frames. In that purpose, code from this notebook is used https://www.kaggle.com/naotaka1128/creating-unnested-dataset","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int64)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float32)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float64)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df\n\nfor file in ['example_test', 'train']:\n    # drop playerTwitterFollowers, teamTwitterFollowers from example_test\n    df = pd.read_csv(f\"{data_dir}/{file}.csv\").dropna(axis=1,how='all')\n    daily_data_nested_df_names = df.drop('date', axis = 1).columns.values.tolist()\n\n    for df_name in daily_data_nested_df_names:\n        date_nested_table = df[['date', df_name]]\n\n        date_nested_table = (date_nested_table[\n          ~pd.isna(date_nested_table[df_name])\n          ].\n          reset_index(drop = True)\n          )\n\n        daily_dfs_collection = []\n\n        for date_index, date_row in date_nested_table.iterrows():\n            daily_df = pd.read_json(date_row[df_name])\n\n            daily_df['dailyDataDate'] = date_row['date']\n\n            daily_dfs_collection = daily_dfs_collection + [daily_df]\n\n        # Concatenate all daily dfs into single df for each row\n        unnested_table = (pd.concat(daily_dfs_collection,\n          ignore_index = True).\n          # Set and reset index to move 'dailyDataDate' to front of df\n          set_index('dailyDataDate').\n          reset_index()\n          )\n        #print(f\"{file}_{df_name}.pickle\")\n        #display(unnested_table.head(3))\n        reduce_mem_usage(unnested_table).to_pickle(f\"{file}_{df_name}.pickle\")\n        #print('\\n'*2)\n\n        # Clean up tables and collection of daily data frames for this df\n        del(date_nested_table, daily_dfs_collection, unnested_table)\n\ndel train","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-06-27T17:17:06.485451Z","iopub.execute_input":"2021-06-27T17:17:06.485843Z","iopub.status.idle":"2021-06-27T17:22:55.985729Z","shell.execute_reply.started":"2021-06-27T17:17:06.485812Z","shell.execute_reply":"2021-06-27T17:22:55.983647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Content:\n\n* [train_nextDayPlayerEngagement.pickle (Target)](#topic1)\n* [players.csv](#topic2)\n* [teams.csv](#topic3)\n* [seasons.csv](#topic4)\n* [awards.csv](#topic5)\n  * [train_awards.pickle](#topic6)\n  * [train_events.pickle](#topic7)\n  * [train_games.pickle](#topic8)\n  * [train_playerBoxScores.pickle](#topic9)\n  * [train_playerTwitterFollowers.pickle](#topic10)\n  * [train_rosters.pickle](#topic11)\n  * [train_standings.pickle](#topic12)\n  * [train_teamBoxScores.pickle](#topic13)\n  * [train_teamTwitterFollowers.pickle](#topic14)\n  * [transactions.pickle](#topic15)","metadata":{}},{"cell_type":"markdown","source":"<a id =topic1> </a>\n# Target","metadata":{}},{"cell_type":"code","source":"train_target = pd.read_pickle('train_nextDayPlayerEngagement.pickle')\ntrain_target['engagementMetricsDate'] = pd.to_datetime(train_target['engagementMetricsDate'])\ntrain_target['dailyDataDate'] = train_target['dailyDataDate'].astype(str)\ntrain_target['dailyDataDate'] = pd.to_datetime(train_target['dailyDataDate'], format=\"%Y%m%d\")\ntrain_target.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:22:55.990655Z","iopub.execute_input":"2021-06-27T17:22:55.991018Z","iopub.status.idle":"2021-06-27T17:23:01.211008Z","shell.execute_reply.started":"2021-06-27T17:22:55.990982Z","shell.execute_reply":"2021-06-27T17:23:01.209849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_target.groupby('engagementMetricsDate').count()['playerId'].plot(figsize=(10,5))\nplt.title('Number of players per date')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:01.212050Z","iopub.execute_input":"2021-06-27T17:23:01.212329Z","iopub.status.idle":"2021-06-27T17:23:01.811951Z","shell.execute_reply.started":"2021-06-27T17:23:01.212301Z","shell.execute_reply":"2021-06-27T17:23:01.810766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 1, figsize=(12,15))\n\nfor i, ax in enumerate(axes):\n    train_target.groupby('engagementMetricsDate').mean()[f'target{i+1}'].plot(ax=ax)\n    ax.set_title(f'mean target{i+1}')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:01.813246Z","iopub.execute_input":"2021-06-27T17:23:01.813591Z","iopub.status.idle":"2021-06-27T17:23:03.507550Z","shell.execute_reply.started":"2021-06-27T17:23:01.813560Z","shell.execute_reply":"2021-06-27T17:23:03.506691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic2> </a>\n# Players","metadata":{}},{"cell_type":"code","source":"players = pd.read_csv(f'{data_dir}/players.csv')\nplayers.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:03.508598Z","iopub.execute_input":"2021-06-27T17:23:03.508927Z","iopub.status.idle":"2021-06-27T17:23:03.545733Z","shell.execute_reply.started":"2021-06-27T17:23:03.508882Z","shell.execute_reply":"2021-06-27T17:23:03.544703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players.groupby('birthCountry').count()['playerId'].sort_values().plot.barh(figsize=(5, 8))\nplt.title('Player birth country')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:03.548261Z","iopub.execute_input":"2021-06-27T17:23:03.548598Z","iopub.status.idle":"2021-06-27T17:23:03.873536Z","shell.execute_reply.started":"2021-06-27T17:23:03.548563Z","shell.execute_reply":"2021-06-27T17:23:03.872422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_temp = players.groupby('primaryPositionName').count()['playerId'].sort_values()\n\ny_pos = np.arange(len(df_temp))\n\nplt.barh(y_pos, df_temp.values, align='center')\nplt.yticks(y_pos, [f'{x}_{y}' for x, y in zip(df_temp.index, df_temp.values)])\nplt.title('Player primary position name')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:03.875379Z","iopub.execute_input":"2021-06-27T17:23:03.875856Z","iopub.status.idle":"2021-06-27T17:23:04.067606Z","shell.execute_reply.started":"2021-06-27T17:23:03.875806Z","shell.execute_reply":"2021-06-27T17:23:04.066875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 1, figsize=(12,15))\nfig.subplots_adjust(hspace=0.3)\n\ndf_agg_temp = train_target.merge(players[['playerId', 'primaryPositionName']], how='left')\nfor pos in df_agg_temp['primaryPositionName'].unique():\n    if pos == 'Pitcher':\n        lw = 3\n        al = 1\n    else:\n        lw = 1\n        al = 0.5\n    for i, ax in enumerate(axes):\n        df_agg_temp[df_agg_temp['primaryPositionName']==pos].groupby(\n            'engagementMetricsDate').mean()[f'target{i+1}'].plot(ax=ax, label=pos,\n                                                                 linewidth=lw, alpha=al, figsize=(17, 30))\n        ax.set_title(f'mean target{i+1}')\nfor i, ax in enumerate(axes):\n    ax.legend(loc=\"upper left\")\n\ndel df_agg_temp, df_temp","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:04.068806Z","iopub.execute_input":"2021-06-27T17:23:04.069297Z","iopub.status.idle":"2021-06-27T17:23:21.512165Z","shell.execute_reply.started":"2021-06-27T17:23:04.069244Z","shell.execute_reply":"2021-06-27T17:23:21.511258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* ### Designated Hitters have highest peaks althought it might be because of the low number of them (only 6).\n* ### Target 2 shows some significant spikes in other positions (like First Base and outfielder).","metadata":{}},{"cell_type":"code","source":"players.groupby('playerForTestSetAndFuturePreds').count()['playerId'].plot.bar()\nplt.title('True if player is among those for whom predictions are to be made in test data')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:21.513415Z","iopub.execute_input":"2021-06-27T17:23:21.513857Z","iopub.status.idle":"2021-06-27T17:23:21.673061Z","shell.execute_reply.started":"2021-06-27T17:23:21.513810Z","shell.execute_reply":"2021-06-27T17:23:21.671917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 1, figsize=(12,15))\nfig.subplots_adjust(hspace=0.3)\n\ndf_agg_temp = train_target.merge(players[['playerId', 'playerForTestSetAndFuturePreds']], how='left')\nfor in_test in [True, False]:\n    for i, ax in enumerate(axes):\n        df_agg_temp[df_agg_temp['playerForTestSetAndFuturePreds']==in_test].groupby(\n            'engagementMetricsDate').mean()[f'target{i+1}'].plot(ax=ax, label=f'Player in test: {in_test}',\n                                                                 figsize=(17, 30))\n        ax.set_title(f'mean target{i+1}')\nfor i, ax in enumerate(axes):\n    ax.legend(loc=\"upper left\")\n\ndel df_agg_temp","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:21.674485Z","iopub.execute_input":"2021-06-27T17:23:21.674787Z","iopub.status.idle":"2021-06-27T17:23:27.241031Z","shell.execute_reply.started":"2021-06-27T17:23:21.674756Z","shell.execute_reply":"2021-06-27T17:23:27.239831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic3> </a>\n# Teams","metadata":{}},{"cell_type":"code","source":"teams = pd.read_csv(f'{data_dir}/teams.csv')\nteams.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:27.242492Z","iopub.execute_input":"2021-06-27T17:23:27.242805Z","iopub.status.idle":"2021-06-27T17:23:27.273381Z","shell.execute_reply.started":"2021-06-27T17:23:27.242766Z","shell.execute_reply":"2021-06-27T17:23:27.272140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_rosters = pd.read_pickle(f'train_rosters.pickle')\ntrain_target = pd.read_pickle('train_nextDayPlayerEngagement.pickle')\nteams_agg = pd.merge(train_target, train_rosters, left_on=['dailyDataDate', 'playerId'],\n                     right_on=['dailyDataDate', 'playerId'], how = 'left')\nteams_agg = pd.merge(teams_agg, teams, left_on=['teamId'], right_on=['id'], how='left')\nfor i in range(4):\n    teams_agg.groupby('shortName').mean()[f'target{i+1}'].sort_values().plot.barh(figsize=(10, 5))\n    plt.xlabel(f'mean target{i+1}')\n    plt.title(f'mean target{i+1} per team')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:27.274871Z","iopub.execute_input":"2021-06-27T17:23:27.275221Z","iopub.status.idle":"2021-06-27T17:23:38.109761Z","shell.execute_reply.started":"2021-06-27T17:23:27.275190Z","shell.execute_reply":"2021-06-27T17:23:38.108603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(4):\n    teams_agg.groupby('leagueName').mean()[f'target{i+1}'].sort_values().plot.barh(figsize=(4, 2))\n    plt.xlabel(f'mean target{i+1}')\n    plt.title(f'mean target{i+1} per league')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:38.111295Z","iopub.execute_input":"2021-06-27T17:23:38.111581Z","iopub.status.idle":"2021-06-27T17:23:41.760058Z","shell.execute_reply.started":"2021-06-27T17:23:38.111552Z","shell.execute_reply":"2021-06-27T17:23:41.758752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(4):\n    teams_agg.groupby('divisionName').mean()[f'target{i+1}'].sort_values().plot.barh(figsize=(10, 5))\n    plt.xlabel(f'mean target{i+1}')\n    plt.title(f'mean target{i+1} per division')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:23:41.761804Z","iopub.execute_input":"2021-06-27T17:23:41.762304Z","iopub.status.idle":"2021-06-27T17:23:45.563259Z","shell.execute_reply.started":"2021-06-27T17:23:41.762250Z","shell.execute_reply":"2021-06-27T17:23:45.562254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic4> </a>\n# Seasons","metadata":{}},{"cell_type":"code","source":"seasons = pd.read_csv(f'{data_dir}/seasons.csv')\nseasons.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:24:55.500118Z","iopub.execute_input":"2021-06-27T17:24:55.500684Z","iopub.status.idle":"2021-06-27T17:24:55.540846Z","shell.execute_reply.started":"2021-06-27T17:24:55.500643Z","shell.execute_reply":"2021-06-27T17:24:55.539934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del seasons","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:00.221563Z","iopub.execute_input":"2021-06-27T17:25:00.222172Z","iopub.status.idle":"2021-06-27T17:25:00.226358Z","shell.execute_reply.started":"2021-06-27T17:25:00.222134Z","shell.execute_reply":"2021-06-27T17:25:00.225212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic5> </a>\n# Awards","metadata":{}},{"cell_type":"code","source":"awards = pd.read_csv(f'{data_dir}/awards.csv')\nawards.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:02.778515Z","iopub.execute_input":"2021-06-27T17:25:02.778923Z","iopub.status.idle":"2021-06-27T17:25:02.828638Z","shell.execute_reply.started":"2021-06-27T17:25:02.778873Z","shell.execute_reply":"2021-06-27T17:25:02.827521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"awards.groupby('playerName').count()['awardId'].sort_values()[-20:].plot.barh()\nplt.title('Top 20 players per number of awards for period 1997-2017')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:07.949555Z","iopub.execute_input":"2021-06-27T17:25:07.949962Z","iopub.status.idle":"2021-06-27T17:25:08.241542Z","shell.execute_reply.started":"2021-06-27T17:25:07.949899Z","shell.execute_reply":"2021-06-27T17:25:08.240365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del awards","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:12.749184Z","iopub.execute_input":"2021-06-27T17:25:12.749697Z","iopub.status.idle":"2021-06-27T17:25:12.753283Z","shell.execute_reply.started":"2021-06-27T17:25:12.749663Z","shell.execute_reply":"2021-06-27T17:25:12.752544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic6> </a>\n# Train awards","metadata":{}},{"cell_type":"code","source":"train_awards = pd.read_pickle(f'train_awards.pickle')\ntrain_awards.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:18.625304Z","iopub.execute_input":"2021-06-27T17:25:18.625824Z","iopub.status.idle":"2021-06-27T17:25:18.660678Z","shell.execute_reply.started":"2021-06-27T17:25:18.625790Z","shell.execute_reply":"2021-06-27T17:25:18.659962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_awards.groupby('playerName').count()['awardId'].sort_values()[-20:].plot.barh()\nplt.title('Top 20 players per number of awards for training period')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:27.079848Z","iopub.execute_input":"2021-06-27T17:25:27.080860Z","iopub.status.idle":"2021-06-27T17:25:27.626959Z","shell.execute_reply.started":"2021-06-27T17:25:27.080802Z","shell.execute_reply":"2021-06-27T17:25:27.625670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_player_targets = train_target[train_target['playerId']==624413]\naward_dates = train_awards[train_awards['playerId'] == 624413]['dailyDataDate'].to_list()\ntop_player_targets['award_date'] = top_player_targets['dailyDataDate'].isin(award_dates).astype(int)\n\nfor i in range(4):\n    top_player_targets[f'target{i+1}'].plot(figsize = (20, 5))\n    top_player_targets[top_player_targets['award_date']==1][f'target{i+1}'].plot(\n        figsize = (20, 5), style='o-',markerfacecolor='red', linestyle='none')\n    plt.legend([f'target{i+1}', 'awards'])\n    plt.title(f'Pete Alonso target{i+1} and awards')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:35.311535Z","iopub.execute_input":"2021-06-27T17:25:35.311963Z","iopub.status.idle":"2021-06-27T17:25:36.551599Z","shell.execute_reply.started":"2021-06-27T17:25:35.311898Z","shell.execute_reply":"2021-06-27T17:25:36.550385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"second_player_targets = train_target[train_target['playerId']==605141]\naward_dates = train_awards[train_awards['playerId'] == 605141]['dailyDataDate'].to_list()\nsecond_player_targets['award_date'] = second_player_targets['dailyDataDate'].isin(award_dates).astype(int)\n\nfor i in range(4):\n    second_player_targets[f'target{i+1}'].plot(figsize = (20, 5))\n    second_player_targets[second_player_targets['award_date']==1][f'target{i+1}'].plot(\n        figsize = (20, 5), style='o-',markerfacecolor='red', linestyle='none')\n    plt.legend([f'target{i+1}', 'awards'])\n    plt.title(f'Wander Franco target{i+1} and awards')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:43.119056Z","iopub.execute_input":"2021-06-27T17:25:43.119449Z","iopub.status.idle":"2021-06-27T17:25:44.056333Z","shell.execute_reply.started":"2021-06-27T17:25:43.119418Z","shell.execute_reply":"2021-06-27T17:25:44.055409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_awards","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:49.072153Z","iopub.execute_input":"2021-06-27T17:25:49.072548Z","iopub.status.idle":"2021-06-27T17:25:49.078499Z","shell.execute_reply.started":"2021-06-27T17:25:49.072516Z","shell.execute_reply":"2021-06-27T17:25:49.077088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic7> </a>\n# Train events","metadata":{}},{"cell_type":"code","source":"train_events = pd.read_pickle('train_events.pickle')\ntrain_events.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:25:54.773025Z","iopub.execute_input":"2021-06-27T17:25:54.773414Z","iopub.status.idle":"2021-06-27T17:26:04.245201Z","shell.execute_reply.started":"2021-06-27T17:25:54.773381Z","shell.execute_reply":"2021-06-27T17:26:04.244039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_col = [col for col in train_events.columns if pd.api.types.is_numeric_dtype(train_events[col])]\nfig, axes = plt.subplots(nrows=18, ncols=3)\nplt.suptitle('Histograms for numeric columns in train_events data frame', y=0.9)\nfig.set_figheight(40)\nfig.set_figwidth(20)\nfig.subplots_adjust(hspace=0.4)\ncolumns = list(train_events.columns)\n\nfor i, ax in enumerate(axes.flatten()):\n    try:\n        train_events[num_col[i]].hist(ax=ax)\n        ax.set_title(num_col[i])\n    except:\n        continue\n        \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:26:04.247085Z","iopub.execute_input":"2021-06-27T17:26:04.247416Z","iopub.status.idle":"2021-06-27T17:26:14.142475Z","shell.execute_reply.started":"2021-06-27T17:26:04.247382Z","shell.execute_reply":"2021-06-27T17:26:14.141222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_col = list(set(train_events.columns).difference(num_col))\nfig, axes = plt.subplots(nrows=10, ncols=2)\nplt.suptitle('Top 20 values for each non numeric columns', y=0.9)\nfig.set_figheight(40)\nfig.set_figwidth(20)\nfig.subplots_adjust(hspace=0.4)\n\nfor i, ax in enumerate(axes.flatten()):\n    \n    try:\n        train_events.groupby(object_col[i]).count()['dailyDataDate'].sort_values()[-20:].plot.barh(ax=ax)\n        ax.set_title(object_col[i])\n    except:\n        continue\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:26:14.144691Z","iopub.execute_input":"2021-06-27T17:26:14.145193Z","iopub.status.idle":"2021-06-27T17:28:03.518062Z","shell.execute_reply.started":"2021-06-27T17:26:14.145139Z","shell.execute_reply":"2021-06-27T17:28:03.516919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_events","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:28:03.519338Z","iopub.execute_input":"2021-06-27T17:28:03.519678Z","iopub.status.idle":"2021-06-27T17:28:03.838311Z","shell.execute_reply.started":"2021-06-27T17:28:03.519644Z","shell.execute_reply":"2021-06-27T17:28:03.837199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic8> </a>\n# Train games","metadata":{}},{"cell_type":"code","source":"train_games = pd.read_pickle('train_games.pickle')\ntrain_games.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:28:12.754059Z","iopub.execute_input":"2021-06-27T17:28:12.754467Z","iopub.status.idle":"2021-06-27T17:28:12.812597Z","shell.execute_reply.started":"2021-06-27T17:28:12.754434Z","shell.execute_reply":"2021-06-27T17:28:12.811555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_col = [col for col in train_games.columns if pd.api.types.is_numeric_dtype(train_games[col])]\nfig, axes = plt.subplots(nrows=7, ncols=3)\nplt.suptitle('Histograms for numeric columns in train_games data frame', y=0.9)\nfig.set_figheight(40)\nfig.set_figwidth(20)\nfig.subplots_adjust(hspace=0.4)\n\nfor i, ax in enumerate(axes.flatten()):\n    \n    try:\n        train_games[num_col[i]].hist(ax=ax)\n        ax.set_title(num_col[i])\n    except:\n        continue\n        \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:28:41.126548Z","iopub.execute_input":"2021-06-27T17:28:41.127012Z","iopub.status.idle":"2021-06-27T17:28:44.455675Z","shell.execute_reply.started":"2021-06-27T17:28:41.126966Z","shell.execute_reply":"2021-06-27T17:28:44.454484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_col = list(set(train_games.columns).difference(num_col))\nfig, axes = plt.subplots(nrows=7, ncols=2, dpi=120)\nplt.suptitle('Top 20 values for each non numeric columns in train_games data frame', y=0.9)\nfig.set_figheight(30)\nfig.set_figwidth(20)\nfig.subplots_adjust(hspace=0.3)\n\nfor i, ax in enumerate(axes.flatten()):\n    \n    try:\n        train_games.groupby(object_col[i]).count()['dailyDataDate'].sort_values()[-20:].plot.barh(ax=ax)\n        ax.set_title(object_col[i])\n    except:\n        continue\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:28:53.500707Z","iopub.execute_input":"2021-06-27T17:28:53.501128Z","iopub.status.idle":"2021-06-27T17:28:57.734656Z","shell.execute_reply.started":"2021-06-27T17:28:53.501091Z","shell.execute_reply":"2021-06-27T17:28:57.733523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_games","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:28:57.736438Z","iopub.execute_input":"2021-06-27T17:28:57.736840Z","iopub.status.idle":"2021-06-27T17:28:57.743030Z","shell.execute_reply.started":"2021-06-27T17:28:57.736801Z","shell.execute_reply":"2021-06-27T17:28:57.741772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic9> </a>\n# Train BoxScores","metadata":{}},{"cell_type":"code","source":"train_playerBoxScores = pd.read_pickle(f'train_playerBoxScores.pickle')\ntrain_playerBoxScores.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:29:16.458957Z","iopub.execute_input":"2021-06-27T17:29:16.459404Z","iopub.status.idle":"2021-06-27T17:29:16.916274Z","shell.execute_reply.started":"2021-06-27T17:29:16.459365Z","shell.execute_reply":"2021-06-27T17:29:16.915146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_col = [col for col in train_playerBoxScores.columns\n           if pd.api.types.is_numeric_dtype(train_playerBoxScores[col])]\nfig, axes = plt.subplots(nrows=27, ncols=3)\nplt.suptitle('Histograms for numeric columns in train_playerBoxScores data frame', y=0.9)\nfig.set_figheight(60)\nfig.set_figwidth(20)\nfig.subplots_adjust(hspace=0.4)\n\nfor i, ax in enumerate(axes.flatten()):\n    \n    try:\n        train_playerBoxScores[num_col[i]].hist(ax=ax)\n        ax.set_title(num_col[i])\n    except:\n        continue\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:29:24.492469Z","iopub.execute_input":"2021-06-27T17:29:24.492864Z","iopub.status.idle":"2021-06-27T17:29:36.531856Z","shell.execute_reply.started":"2021-06-27T17:29:24.492830Z","shell.execute_reply":"2021-06-27T17:29:36.530550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_col = list(set(train_playerBoxScores.columns).difference(num_col))\nfig, axes = plt.subplots(nrows=4, ncols=2, dpi=120)\nplt.suptitle('Top 20 values for each non numeric columns', y=0.9)\nfig.set_figheight(20)\nfig.set_figwidth(20)\nfig.subplots_adjust(hspace=0.3)\nfor i, ax in enumerate(axes.flatten()):\n    \n    try:\n        train_playerBoxScores.groupby(object_col[i]).count()['dailyDataDate'].sort_values()[-20:].plot.barh(ax=ax)\n        ax.set_title(object_col[i])\n    except:\n        continue\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:30:15.186131Z","iopub.execute_input":"2021-06-27T17:30:15.186792Z","iopub.status.idle":"2021-06-27T17:30:18.891457Z","shell.execute_reply.started":"2021-06-27T17:30:15.186746Z","shell.execute_reply":"2021-06-27T17:30:18.890561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_playerBoxScores","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:30:21.516573Z","iopub.execute_input":"2021-06-27T17:30:21.517188Z","iopub.status.idle":"2021-06-27T17:30:21.531866Z","shell.execute_reply.started":"2021-06-27T17:30:21.517152Z","shell.execute_reply":"2021-06-27T17:30:21.530362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic10> </a>\n# Train Player Twitter Followers","metadata":{}},{"cell_type":"code","source":"train_playerTwitterFollowers = pd.read_pickle(f'train_playerTwitterFollowers.pickle')\ntrain_playerTwitterFollowers.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:31:56.564273Z","iopub.execute_input":"2021-06-27T17:31:56.564834Z","iopub.status.idle":"2021-06-27T17:31:56.605472Z","shell.execute_reply.started":"2021-06-27T17:31:56.564797Z","shell.execute_reply":"2021-06-27T17:31:56.604651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_playerTwitterFollowers.groupby('playerName').max()['numberOfFollowers'].sort_values()[-20:].plot.barh()\nplt.title('Top 20 players with the most twitter followers')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:32:00.343971Z","iopub.execute_input":"2021-06-27T17:32:00.344569Z","iopub.status.idle":"2021-06-27T17:32:01.017964Z","shell.execute_reply.started":"2021-06-27T17:32:00.344529Z","shell.execute_reply":"2021-06-27T17:32:01.015821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_playerTwitterFollowers","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:32:10.954109Z","iopub.execute_input":"2021-06-27T17:32:10.954542Z","iopub.status.idle":"2021-06-27T17:32:10.959829Z","shell.execute_reply.started":"2021-06-27T17:32:10.954492Z","shell.execute_reply":"2021-06-27T17:32:10.958625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic11> </a>\n# Train rosters","metadata":{}},{"cell_type":"code","source":"train_rosters = pd.read_pickle('train_rosters.pickle')\ntrain_rosters.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:32:58.341785Z","iopub.execute_input":"2021-06-27T17:32:58.342304Z","iopub.status.idle":"2021-06-27T17:32:59.344633Z","shell.execute_reply.started":"2021-06-27T17:32:58.342259Z","shell.execute_reply":"2021-06-27T17:32:59.343813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_rosters.groupby('playerId')['teamId'].nunique().hist()\nplt.title('The number of different teams that the player changed')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:33:08.581648Z","iopub.execute_input":"2021-06-27T17:33:08.582155Z","iopub.status.idle":"2021-06-27T17:33:09.557979Z","shell.execute_reply.started":"2021-06-27T17:33:08.582112Z","shell.execute_reply":"2021-06-27T17:33:09.557010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_rosters.merge(players[['playerId', 'playerName']], left_on=['playerId'],\n                    right_on=['playerId'], how='left').groupby(\n    'playerName')['teamId'].nunique().sort_values()[-40:].plot.barh(figsize=(5,10))\nplt.title('Players who changed the most different teams')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:33:30.597215Z","iopub.execute_input":"2021-06-27T17:33:30.597759Z","iopub.status.idle":"2021-06-27T17:33:31.950883Z","shell.execute_reply.started":"2021-06-27T17:33:30.597707Z","shell.execute_reply":"2021-06-27T17:33:31.949839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_rosters","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:33:42.216794Z","iopub.execute_input":"2021-06-27T17:33:42.217231Z","iopub.status.idle":"2021-06-27T17:33:42.222598Z","shell.execute_reply.started":"2021-06-27T17:33:42.217193Z","shell.execute_reply":"2021-06-27T17:33:42.221226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic12> </a>\n# Train standings","metadata":{}},{"cell_type":"code","source":"train_standings = pd.read_pickle('train_standings.pickle')\ntrain_standings.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:34:26.138977Z","iopub.execute_input":"2021-06-27T17:34:26.139397Z","iopub.status.idle":"2021-06-27T17:34:26.210772Z","shell.execute_reply.started":"2021-06-27T17:34:26.139356Z","shell.execute_reply":"2021-06-27T17:34:26.209936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_rosters = pd.read_pickle('train_rosters.pickle')\ndf_temp = pd.merge(train_target, train_rosters, left_on=['dailyDataDate', 'playerId'],\n                   right_on=['dailyDataDate', 'playerId'], how='left')\n\nfor col in ['engagementMetricsDate', 'gameDate', 'status', 'statusCode']:\n    df_temp = df_temp.drop(col, axis=1)\n\ndf_temp = pd.merge(df_temp, train_standings, left_on=['dailyDataDate', 'teamId'],\n                   right_on=['dailyDataDate', 'teamId'], how='left')\n\ndf_corr = df_temp.corr()\nplt.rcParams[\"figure.figsize\"] = (17,17)\nsns.heatmap(df_corr, xticklabels=df_corr.columns, yticklabels=df_corr.columns, annot=True)\nplt.title('Corerlation between tagret columns and columns from standing data frame')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:35:33.606813Z","iopub.execute_input":"2021-06-27T17:35:33.608187Z","iopub.status.idle":"2021-06-27T17:36:03.041748Z","shell.execute_reply.started":"2021-06-27T17:35:33.608147Z","shell.execute_reply":"2021-06-27T17:36:03.040589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_rosters, train_standings","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:36:07.052944Z","iopub.execute_input":"2021-06-27T17:36:07.053357Z","iopub.status.idle":"2021-06-27T17:36:07.159005Z","shell.execute_reply.started":"2021-06-27T17:36:07.053311Z","shell.execute_reply":"2021-06-27T17:36:07.157175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic13> </a>\n# Train teamBoxScores","metadata":{}},{"cell_type":"code","source":"train_teamBoxScores = pd.read_pickle('train_teamBoxScores.pickle')\ntrain_teamBoxScores.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:37:11.280502Z","iopub.execute_input":"2021-06-27T17:37:11.280978Z","iopub.status.idle":"2021-06-27T17:37:11.338189Z","shell.execute_reply.started":"2021-06-27T17:37:11.280928Z","shell.execute_reply":"2021-06-27T17:37:11.337151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_rosters = pd.read_pickle('train_rosters.pickle')\ndf_temp = pd.merge(train_target, train_rosters, left_on=['dailyDataDate', 'playerId'],\n                   right_on=['dailyDataDate', 'playerId'], how='left')\n\nfor col in ['engagementMetricsDate', 'gameDate', 'status', 'statusCode']:\n    df_temp = df_temp.drop(col, axis=1)\n\ndf_temp = pd.merge(df_temp, train_teamBoxScores, left_on=['dailyDataDate', 'teamId'],\n                   right_on=['dailyDataDate', 'teamId'], how='left')\n\ndf_corr = df_temp.corr()\nplt.rcParams[\"figure.figsize\"] = (17,17)\nsns.heatmap(df_corr, xticklabels=df_corr.columns, yticklabels=df_corr.columns, annot=True)\nplt.title('Corerlation between tagret columns and columns from teamBoxScores data frame')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:37:53.345853Z","iopub.execute_input":"2021-06-27T17:37:53.346257Z","iopub.status.idle":"2021-06-27T17:38:30.725287Z","shell.execute_reply.started":"2021-06-27T17:37:53.346219Z","shell.execute_reply":"2021-06-27T17:38:30.724368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_rosters, df_temp, train_teamBoxScores","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:38:30.726692Z","iopub.execute_input":"2021-06-27T17:38:30.727241Z","iopub.status.idle":"2021-06-27T17:38:30.858620Z","shell.execute_reply.started":"2021-06-27T17:38:30.727201Z","shell.execute_reply":"2021-06-27T17:38:30.857368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic14> </a>\n# Train teamTwitterFollowers","metadata":{}},{"cell_type":"code","source":"train_teamTwitterFollowers = pd.read_pickle('train_teamTwitterFollowers.pickle')\ntrain_teamTwitterFollowers.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:39:15.626703Z","iopub.execute_input":"2021-06-27T17:39:15.627161Z","iopub.status.idle":"2021-06-27T17:39:15.649682Z","shell.execute_reply.started":"2021-06-27T17:39:15.627118Z","shell.execute_reply":"2021-06-27T17:39:15.648601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams[\"figure.figsize\"] = (7,7)\ntrain_teamTwitterFollowers.groupby('teamName').max()['numberOfFollowers'].sort_values()[-20:].plot.barh()\nplt.title('Top 20 teams with the most twitter followers')","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:39:27.481142Z","iopub.execute_input":"2021-06-27T17:39:27.481557Z","iopub.status.idle":"2021-06-27T17:39:27.796022Z","shell.execute_reply.started":"2021-06-27T17:39:27.481519Z","shell.execute_reply":"2021-06-27T17:39:27.794974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_teamTwitterFollowers","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:39:46.854042Z","iopub.execute_input":"2021-06-27T17:39:46.854476Z","iopub.status.idle":"2021-06-27T17:39:46.859303Z","shell.execute_reply.started":"2021-06-27T17:39:46.854439Z","shell.execute_reply":"2021-06-27T17:39:46.857835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id =topic15> </a>\n# Train transactions","metadata":{}},{"cell_type":"code","source":"train_transactions = pd.read_pickle('train_transactions.pickle')\ntrain_transactions.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-27T17:40:49.950032Z","iopub.execute_input":"2021-06-27T17:40:49.950441Z","iopub.status.idle":"2021-06-27T17:40:50.039562Z","shell.execute_reply.started":"2021-06-27T17:40:49.950406Z","shell.execute_reply":"2021-06-27T17:40:50.038232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}