{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport ipywidgets as widgets\nfrom sklearn.preprocessing import StandardScaler\nimport gc\nfrom pathlib import Path","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:52:00.438292Z","iopub.status.busy":"2021-07-22T13:52:00.437045Z","iopub.status.idle":"2021-07-22T13:52:01.574146Z","shell.execute_reply":"2021-07-22T13:52:01.573423Z","shell.execute_reply.started":"2021-07-22T13:12:33.048598Z"},"papermill":{"duration":1.185329,"end_time":"2021-07-22T13:52:01.574355","exception":false,"start_time":"2021-07-22T13:52:00.389026","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Start with input file path\ninput_file_path = Path('/kaggle/input/mlb-player-digital-engagement-forecasting/')\n\n\n# Create table with list of CSV files to be read in, w/ corresponding df name\n# This does include large 'train' data set (read in separately)\ncsv_and_df_names = pd.DataFrame(data = {\n  'csv_name': ['seasons', 'teams', 'players', 'awards'],\n  'df_name': ['seasons', 'teams', 'players', 'awards_pre2018'] \n  })\n\n# Set up for tabbed output\nkaggle_data_tabs = widgets.Tab()\n\n# Add Output widgets for each (eventual) DF as tabs' children\nkaggle_data_tabs.children = list([widgets.Output() for df_name \n  in csv_and_df_names['df_name']])\n\nfor index, row in csv_and_df_names.iterrows():\n    \n    csv_name = row['csv_name']\n    df_name = row['df_name']\n    \n    # Read from CSV and create df with specified name in environment\n    globals()[df_name] = pd.read_csv(input_file_path / f\"{csv_name}.csv\")\n\n    # Set tab title to df name\n    kaggle_data_tabs.set_title(index, df_name)\n    \n    # Display corresponding table output for this tab name\n    with kaggle_data_tabs.children[index]:\n        display(eval(df_name))\n\ndisplay(kaggle_data_tabs)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:52:01.706172Z","iopub.status.busy":"2021-07-22T13:52:01.705051Z","iopub.status.idle":"2021-07-22T13:52:02.012093Z","shell.execute_reply":"2021-07-22T13:52:02.010002Z","shell.execute_reply.started":"2021-07-22T13:12:35.571274Z"},"papermill":{"duration":0.385801,"end_time":"2021-07-22T13:52:02.012272","exception":false,"start_time":"2021-07-22T13:52:01.626471","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(input_file_path /'train.csv')\n\n# Convert training data date field to pandas datetime type\ntrain['date'] = pd.to_datetime(train['date'], format = \"%Y%m%d\")\n\ndisplay(train.info())\n\ndisplay(train)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:52:02.108539Z","iopub.status.busy":"2021-07-22T13:52:02.107922Z","iopub.status.idle":"2021-07-22T13:53:28.565401Z","shell.execute_reply":"2021-07-22T13:53:28.564847Z","shell.execute_reply.started":"2021-07-22T13:12:41.099894Z"},"papermill":{"duration":86.506766,"end_time":"2021-07-22T13:53:28.565561","exception":false,"start_time":"2021-07-22T13:52:02.058795","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get names of all \"nested\" data frames in daily training set\ndaily_data_nested_df_names = train.drop('date', axis = 1).columns.values.tolist()\n\nfor df_name in daily_data_nested_df_names:\n    date_nested_table = train[['date', df_name]]\n\n    date_nested_table = (date_nested_table[\n      ~pd.isna(date_nested_table[df_name])\n      ].\n      reset_index(drop = True)\n      )\n    \n    daily_dfs_collection = []\n    \n    for date_index, date_row in date_nested_table.iterrows():\n        daily_df = pd.read_json(date_row[df_name])\n        \n        daily_df['dailyDataDate'] = date_row['date']\n        \n        daily_dfs_collection = daily_dfs_collection + [daily_df]\n\n    # Concatenate all daily dfs into single df for each row\n    unnested_table = (pd.concat(daily_dfs_collection,\n      ignore_index = True).\n      # Set and reset index to move 'dailyDataDate' to front of df\n      set_index('dailyDataDate').\n      reset_index()\n      )\n    \n    # Creates 1 pandas df per unnested df from daily data read in, with same name\n    globals()[df_name] = unnested_table    \n    \n    # Clean up tables and collection of daily data frames for this df\n    del(date_nested_table, daily_dfs_collection, unnested_table)\n\n# Set up for tabbed output\ndaily_data_unnested_tabs = widgets.Tab()\n\n# Add Output widgets for each (eventual) DF as tabs' children\ndaily_data_unnested_tabs.children = list([widgets.Output() \n  for df_name in daily_data_nested_df_names])\n\nfor index in range(0, len(daily_data_nested_df_names)):\n    df_name = daily_data_nested_df_names[index]\n    \n    # Rename tab bar titles to df names\n    daily_data_unnested_tabs.set_title(index, df_name)\n\n    # Display corresponding table output for this tab name\n    with daily_data_unnested_tabs.children[index]:\n        display(eval(df_name))\n\ndisplay(daily_data_unnested_tabs)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:53:28.67117Z","iopub.status.busy":"2021-07-22T13:53:28.670528Z","iopub.status.idle":"2021-07-22T13:58:43.409634Z","shell.execute_reply":"2021-07-22T13:58:43.407124Z","shell.execute_reply.started":"2021-07-22T13:14:09.015673Z"},"papermill":{"duration":314.794271,"end_time":"2021-07-22T13:58:43.409881","exception":false,"start_time":"2021-07-22T13:53:28.61561","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(train)\n\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:43.519443Z","iopub.status.busy":"2021-07-22T13:58:43.518826Z","iopub.status.idle":"2021-07-22T13:58:43.74244Z","shell.execute_reply":"2021-07-22T13:58:43.742894Z","shell.execute_reply.started":"2021-07-22T13:19:14.300851Z"},"papermill":{"duration":0.283353,"end_time":"2021-07-22T13:58:43.743072","exception":false,"start_time":"2021-07-22T13:58:43.459719","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation between targets**","metadata":{"papermill":{"duration":0.047098,"end_time":"2021-07-22T13:58:43.837595","exception":false,"start_time":"2021-07-22T13:58:43.790497","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(8,6))\ncor = nextDayPlayerEngagement[['target1','target2','target3','target4']].corr()\nsns.heatmap(cor, annot=True, cmap=plt.cm.Reds)\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:43.934184Z","iopub.status.busy":"2021-07-22T13:58:43.9335Z","iopub.status.idle":"2021-07-22T13:58:44.491075Z","shell.execute_reply":"2021-07-22T13:58:44.49054Z","shell.execute_reply.started":"2021-07-22T13:19:14.563543Z"},"papermill":{"duration":0.607084,"end_time":"2021-07-22T13:58:44.491251","exception":false,"start_time":"2021-07-22T13:58:43.884167","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Relationship between important player-level stats with our target**","metadata":{"papermill":{"duration":0.047856,"end_time":"2021-07-22T13:58:44.587317","exception":false,"start_time":"2021-07-22T13:58:44.539461","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"*Compute averagre of the four targets*","metadata":{"papermill":{"duration":0.04742,"end_time":"2021-07-22T13:58:44.682416","exception":false,"start_time":"2021-07-22T13:58:44.634996","status":"completed"},"tags":[]}},{"cell_type":"code","source":"player_eng_info = nextDayPlayerEngagement.copy()\nplayer_eng_info['target1To4Avg'] = np.mean(\n  player_eng_info[['target1', 'target2', 'target3', 'target4']],\n  axis = 1)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:44.785701Z","iopub.status.busy":"2021-07-22T13:58:44.785005Z","iopub.status.idle":"2021-07-22T13:58:45.004337Z","shell.execute_reply":"2021-07-22T13:58:45.003647Z","shell.execute_reply.started":"2021-07-22T13:19:15.114826Z"},"papermill":{"duration":0.273018,"end_time":"2021-07-22T13:58:45.004502","exception":false,"start_time":"2021-07-22T13:58:44.731484","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_eng_info = player_eng_info[player_eng_info['dailyDataDate'] >='2018-03-29']\nplayer_eng_info = pd.merge(\n  player_eng_info,\n  playerBoxScores[['dailyDataDate','playerId','gamePk','teamId', 'playerName', 'runsScored', 'atBats', 'homeRuns','flyOuts','hits','strikes',\n                   'balks','errors','chances','rbi']],\n   on = ['dailyDataDate','playerId'],\n   how = 'inner'\n   )\n","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:45.110855Z","iopub.status.busy":"2021-07-22T13:58:45.109643Z","iopub.status.idle":"2021-07-22T13:58:46.123816Z","shell.execute_reply":"2021-07-22T13:58:46.123103Z","shell.execute_reply.started":"2021-07-22T13:19:15.283867Z"},"papermill":{"duration":1.069973,"end_time":"2021-07-22T13:58:46.123966","exception":false,"start_time":"2021-07-22T13:58:45.053993","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_eng_info.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:46.255226Z","iopub.status.busy":"2021-07-22T13:58:46.250603Z","iopub.status.idle":"2021-07-22T13:58:46.260273Z","shell.execute_reply":"2021-07-22T13:58:46.259732Z","shell.execute_reply.started":"2021-07-22T13:19:16.161721Z"},"papermill":{"duration":0.089008,"end_time":"2021-07-22T13:58:46.260421","exception":false,"start_time":"2021-07-22T13:58:46.171413","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation matrix between them**","metadata":{"papermill":{"duration":0.048223,"end_time":"2021-07-22T13:58:46.357228","exception":false,"start_time":"2021-07-22T13:58:46.309005","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\ncor = player_eng_info[['target1To4Avg','runsScored','atBats','homeRuns','flyOuts','hits','strikes',\n                   'balks','errors','chances','rbi']].corr()\nsns.heatmap(cor, annot=True, cmap=plt.cm.Reds)\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:46.46543Z","iopub.status.busy":"2021-07-22T13:58:46.463142Z","iopub.status.idle":"2021-07-22T13:58:47.46968Z","shell.execute_reply":"2021-07-22T13:58:47.470211Z","shell.execute_reply.started":"2021-07-22T13:19:16.20084Z"},"papermill":{"duration":1.064917,"end_time":"2021-07-22T13:58:47.470411","exception":false,"start_time":"2021-07-22T13:58:46.405494","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*note : There are some features highly correlated with each other and we may choose only one from them and neglect the other*","metadata":{"papermill":{"duration":0.05098,"end_time":"2021-07-22T13:58:47.573096","exception":false,"start_time":"2021-07-22T13:58:47.522116","status":"completed"},"tags":[]}},{"cell_type":"code","source":"standings_stats = pd.merge(\n  player_eng_info,\n  standings[['dailyDataDate','teamId','wins','losses','pct','xWinLossPct','divisionRank',\n             'leagueRank','wildCardRank','lastTenWins','lastTenLosses']],\n   on = ['dailyDataDate','teamId'],\n   how = 'inner'\n   )","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:47.680906Z","iopub.status.busy":"2021-07-22T13:58:47.680321Z","iopub.status.idle":"2021-07-22T13:58:47.833593Z","shell.execute_reply":"2021-07-22T13:58:47.834075Z","shell.execute_reply.started":"2021-07-22T13:19:17.218477Z"},"papermill":{"duration":0.207531,"end_time":"2021-07-22T13:58:47.834293","exception":false,"start_time":"2021-07-22T13:58:47.626762","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation between important stats from standings data and target**","metadata":{"papermill":{"duration":0.050558,"end_time":"2021-07-22T13:58:47.936664","exception":false,"start_time":"2021-07-22T13:58:47.886106","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(8,6))\ncor = standings_stats[['target1To4Avg','wins','losses','pct','xWinLossPct','divisionRank',\n                       'leagueRank','wildCardRank','lastTenWins','lastTenLosses']].corr()\nsns.heatmap(cor, annot=True, cmap=plt.cm.Reds)\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:48.048934Z","iopub.status.busy":"2021-07-22T13:58:48.048131Z","iopub.status.idle":"2021-07-22T13:58:48.987941Z","shell.execute_reply":"2021-07-22T13:58:48.988487Z","shell.execute_reply.started":"2021-07-22T13:19:17.360338Z"},"papermill":{"duration":1.000678,"end_time":"2021-07-22T13:58:48.988663","exception":false,"start_time":"2021-07-22T13:58:47.987985","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data = standings_stats.groupby('dailyDataDate', as_index=True)[['target1To4Avg','runsScored','atBats','homeRuns','flyOuts','hits','strikes',\n                   'balks','errors','chances','rbi','pct','xWinLossPct','divisionRank','leagueRank','wildCardRank','lastTenWins','lastTenLosses']].mean()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:49.116602Z","iopub.status.busy":"2021-07-22T13:58:49.115509Z","iopub.status.idle":"2021-07-22T13:58:49.137538Z","shell.execute_reply":"2021-07-22T13:58:49.136919Z","shell.execute_reply.started":"2021-07-22T13:19:18.2919Z"},"papermill":{"duration":0.093732,"end_time":"2021-07-22T13:58:49.137703","exception":false,"start_time":"2021-07-22T13:58:49.043971","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:49.270316Z","iopub.status.busy":"2021-07-22T13:58:49.269306Z","iopub.status.idle":"2021-07-22T13:58:49.273686Z","shell.execute_reply":"2021-07-22T13:58:49.274113Z","shell.execute_reply.started":"2021-07-22T13:19:18.328195Z"},"papermill":{"duration":0.083171,"end_time":"2021-07-22T13:58:49.274313","exception":false,"start_time":"2021-07-22T13:58:49.191142","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plot Average of targets of all players for each day across time and compare it with average of strikes**","metadata":{"papermill":{"duration":0.053913,"end_time":"2021-07-22T13:58:49.382191","exception":false,"start_time":"2021-07-22T13:58:49.328278","status":"completed"},"tags":[]}},{"cell_type":"code","source":"averaged_data[['target1To4Avg','strikes']].plot()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:49.496271Z","iopub.status.busy":"2021-07-22T13:58:49.4955Z","iopub.status.idle":"2021-07-22T13:58:49.79797Z","shell.execute_reply":"2021-07-22T13:58:49.798536Z","shell.execute_reply.started":"2021-07-22T13:19:18.357623Z"},"papermill":{"duration":0.362636,"end_time":"2021-07-22T13:58:49.798727","exception":false,"start_time":"2021-07-22T13:58:49.436091","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*note : There is some similarity in their behavior across time*","metadata":{"papermill":{"duration":0.056066,"end_time":"2021-07-22T13:58:49.911427","exception":false,"start_time":"2021-07-22T13:58:49.855361","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"**Plot Average of targets of all players for each day across time and compare it with average of atBats**","metadata":{"papermill":{"duration":0.056493,"end_time":"2021-07-22T13:58:50.024961","exception":false,"start_time":"2021-07-22T13:58:49.968468","status":"completed"},"tags":[]}},{"cell_type":"code","source":"averaged_data['atBats'] = averaged_data['atBats']*10","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:50.14181Z","iopub.status.busy":"2021-07-22T13:58:50.140872Z","iopub.status.idle":"2021-07-22T13:58:50.146457Z","shell.execute_reply":"2021-07-22T13:58:50.146977Z","shell.execute_reply.started":"2021-07-22T13:19:18.668245Z"},"papermill":{"duration":0.065795,"end_time":"2021-07-22T13:58:50.14715","exception":false,"start_time":"2021-07-22T13:58:50.081355","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data[['target1To4Avg','atBats']].plot()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:50.27733Z","iopub.status.busy":"2021-07-22T13:58:50.268151Z","iopub.status.idle":"2021-07-22T13:58:50.546097Z","shell.execute_reply":"2021-07-22T13:58:50.54555Z","shell.execute_reply.started":"2021-07-22T13:19:18.676087Z"},"papermill":{"duration":0.343956,"end_time":"2021-07-22T13:58:50.546256","exception":false,"start_time":"2021-07-22T13:58:50.2023","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*note : There is some similarity in their behavior across time*","metadata":{"papermill":{"duration":0.058348,"end_time":"2021-07-22T13:58:50.661731","exception":false,"start_time":"2021-07-22T13:58:50.603383","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"**Plot Average of targets of all players for each day across time and compare it with average of xWinLossPct**","metadata":{"papermill":{"duration":0.057427,"end_time":"2021-07-22T13:58:50.775995","exception":false,"start_time":"2021-07-22T13:58:50.718568","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sns.distplot(averaged_data['xWinLossPct'])\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:50.897016Z","iopub.status.busy":"2021-07-22T13:58:50.896373Z","iopub.status.idle":"2021-07-22T13:58:51.109877Z","shell.execute_reply":"2021-07-22T13:58:51.10925Z","shell.execute_reply.started":"2021-07-22T13:19:18.961898Z"},"papermill":{"duration":0.275308,"end_time":"2021-07-22T13:58:51.11002","exception":false,"start_time":"2021-07-22T13:58:50.834712","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data['xWinLossPct'] = averaged_data['xWinLossPct']*50","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:51.23441Z","iopub.status.busy":"2021-07-22T13:58:51.233727Z","iopub.status.idle":"2021-07-22T13:58:51.236538Z","shell.execute_reply":"2021-07-22T13:58:51.235913Z","shell.execute_reply.started":"2021-07-22T13:19:19.172857Z"},"papermill":{"duration":0.067413,"end_time":"2021-07-22T13:58:51.236773","exception":false,"start_time":"2021-07-22T13:58:51.16936","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data[['target1To4Avg','xWinLossPct']].plot()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:51.384401Z","iopub.status.busy":"2021-07-22T13:58:51.378638Z","iopub.status.idle":"2021-07-22T13:58:51.634494Z","shell.execute_reply":"2021-07-22T13:58:51.633914Z","shell.execute_reply.started":"2021-07-22T13:19:19.181935Z"},"papermill":{"duration":0.340031,"end_time":"2021-07-22T13:58:51.634643","exception":false,"start_time":"2021-07-22T13:58:51.294612","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*note : There is some similarity in their behavior across time*","metadata":{"papermill":{"duration":0.059491,"end_time":"2021-07-22T13:58:51.753342","exception":false,"start_time":"2021-07-22T13:58:51.693851","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\ncor = averaged_data[['target1To4Avg','runsScored','atBats','homeRuns','hits',\n        'balks','chances','rbi','pct','xWinLossPct','divisionRank','leagueRank','wildCardRank','lastTenWins','lastTenLosses']].corr()\nsns.heatmap(cor, annot=True, cmap=plt.cm.Reds)\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:51.881612Z","iopub.status.busy":"2021-07-22T13:58:51.880914Z","iopub.status.idle":"2021-07-22T13:58:53.449288Z","shell.execute_reply":"2021-07-22T13:58:53.449793Z","shell.execute_reply.started":"2021-07-22T13:19:19.460918Z"},"papermill":{"duration":1.637113,"end_time":"2021-07-22T13:58:53.449962","exception":false,"start_time":"2021-07-22T13:58:51.812849","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We see here that 'pct' variable has strong positive correlation with target variable after computing averages of them per every unique day**\n\n**'lastTenWins' variable has strong positive correlation with target and of course 'lastTenLosses' has negative correlation with target also**\n\n**Also 'divisionRank' variable has strong negative correlation with target**\n\n**'leagueRank' variable has strong negative correlation with target**\n\n**'wildCardRank' variable has strong negative correlation with target**\n\n\n*Before grouping data by unique days and computing averages of variables, the correlation between target and 'pct' was only **0.16** and was **-0.16** with 'divisionRank'*","metadata":{"papermill":{"duration":0.06344,"end_time":"2021-07-22T13:58:53.576999","exception":false,"start_time":"2021-07-22T13:58:53.513559","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"**We can also see here that 'divisionRank' variable has completely opposite behavior compared to target across time **","metadata":{"papermill":{"duration":0.063288,"end_time":"2021-07-22T13:58:53.703741","exception":false,"start_time":"2021-07-22T13:58:53.640453","status":"completed"},"tags":[]}},{"cell_type":"code","source":"averaged_data['divisionRank'] = averaged_data['divisionRank']*10","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:53.838094Z","iopub.status.busy":"2021-07-22T13:58:53.837445Z","iopub.status.idle":"2021-07-22T13:58:53.840781Z","shell.execute_reply":"2021-07-22T13:58:53.840094Z","shell.execute_reply.started":"2021-07-22T13:19:20.985587Z"},"papermill":{"duration":0.072924,"end_time":"2021-07-22T13:58:53.840921","exception":false,"start_time":"2021-07-22T13:58:53.767997","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data[['target1To4Avg','divisionRank']].plot()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:54.004338Z","iopub.status.busy":"2021-07-22T13:58:53.995911Z","iopub.status.idle":"2021-07-22T13:58:54.264678Z","shell.execute_reply":"2021-07-22T13:58:54.264083Z","shell.execute_reply.started":"2021-07-22T13:19:20.992605Z"},"papermill":{"duration":0.359475,"end_time":"2021-07-22T13:58:54.264836","exception":false,"start_time":"2021-07-22T13:58:53.905361","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**And here we see that 'Current winning percentage' or 'pct' variable has almost the same behavior as target**","metadata":{"papermill":{"duration":0.065883,"end_time":"2021-07-22T13:58:54.396745","exception":false,"start_time":"2021-07-22T13:58:54.330862","status":"completed"},"tags":[]}},{"cell_type":"code","source":"averaged_data['pct'] = averaged_data['pct']*50","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:54.533567Z","iopub.status.busy":"2021-07-22T13:58:54.532764Z","iopub.status.idle":"2021-07-22T13:58:54.53486Z","shell.execute_reply":"2021-07-22T13:58:54.53537Z","shell.execute_reply.started":"2021-07-22T13:19:21.286868Z"},"papermill":{"duration":0.073796,"end_time":"2021-07-22T13:58:54.53555","exception":false,"start_time":"2021-07-22T13:58:54.461754","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"averaged_data[['target1To4Avg','pct']].plot()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:54.668129Z","iopub.status.busy":"2021-07-22T13:58:54.667518Z","iopub.status.idle":"2021-07-22T13:58:54.94365Z","shell.execute_reply":"2021-07-22T13:58:54.94425Z","shell.execute_reply.started":"2021-07-22T13:19:21.296174Z"},"papermill":{"duration":0.34406,"end_time":"2021-07-22T13:58:54.944446","exception":false,"start_time":"2021-07-22T13:58:54.600386","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data = standings_stats.drop(['engagementMetricsDate','target1To4Avg'],axis=1)\n\nfinal_data = pd.merge(\n  final_data,\n  awards[['dailyDataDate','awardName','playerId']],\n   on = ['dailyDataDate','playerId'],\n   how = 'left'\n   )\n\nfinal_data = final_data.fillna(0)\n\nfinal_data['awardName'] = [1 if x!=0 else 0 for x in final_data['awardName'] ]\n\nfinal_data = pd.merge(\n  final_data,\n  playerTwitterFollowers[['dailyDataDate','numberOfFollowers','playerId']],\n   on = ['dailyDataDate','playerId'],\n   how = 'left'\n   )\n\nfinal_data = pd.merge(\n  final_data,\n  teamTwitterFollowers[['dailyDataDate','numberOfFollowers','teamId']],\n   on = ['dailyDataDate','teamId'],\n   how = 'left'\n   )\n\nfinal_data = pd.merge(\n  final_data,\n  games[['dailyDataDate','gamePk','isTie','gamesInSeries','seriesDescription','homeWinPct','awayWinPct','homeId','awayId']],\n   on = ['dailyDataDate','gamePk'],\n   how = 'left'\n   )\n\nfinal_data['pct_diff'] = (final_data['homeWinPct'] - final_data['awayWinPct']).abs()\n\nfinal_data = pd.merge(\n  final_data,\n  standings[['dailyDataDate','teamId','divisionRank', 'leagueRank','wildCardRank']],\n   left_on = ['dailyDataDate','homeId'],\n   right_on= ['dailyDataDate','teamId'],\n   how = 'left'\n   )\n\nfinal_data = pd.merge(\n  final_data,\n  standings[['dailyDataDate','teamId','divisionRank', 'leagueRank','wildCardRank']],\n   left_on = ['dailyDataDate','awayId'],\n   right_on= ['dailyDataDate','teamId'],\n   how = 'left'\n   )\n\nfinal_data = final_data.rename(columns={'divisionRank_x': 'player_divisionRank', 'divisionRank_y': 'home_divisionRank', \n                                       'divisionRank' : 'away_divisionRank'})\nfinal_data = final_data.rename(columns={'leagueRank_x': 'player_leagueRank', 'leagueRank_y': 'home_leagueRank', \n                                       'leagueRank' : 'away_leagueRank'})\nfinal_data = final_data.rename(columns={'wildCardRank_x': 'player_wildCardRank', 'wildCardRank_y': 'home_wildCardRank', \n                                       'wildCardRank' : 'away_wildCardRank'})\n\nfinal_data['divisionRank_diff'] = (final_data['home_divisionRank'] - final_data['away_divisionRank']).abs()\nfinal_data['leagueRank_diff'] = (final_data['home_leagueRank'] - final_data['away_leagueRank']).abs()\nfinal_data['wildCardRank_diff'] = (final_data['home_wildCardRank'] - final_data['away_wildCardRank']).abs()\n\nfinal_data = final_data.drop(['playerName','homeId', 'awayId'],axis=1)\nfinal_data = final_data.drop(['seriesDescription'],axis=1)\nfinal_data = final_data.drop(['teamId_x','teamId_y', 'teamId'],axis=1)\n","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:55.082779Z","iopub.status.busy":"2021-07-22T13:58:55.08207Z","iopub.status.idle":"2021-07-22T13:58:56.215499Z","shell.execute_reply":"2021-07-22T13:58:56.214717Z","shell.execute_reply.started":"2021-07-22T13:19:21.583264Z"},"papermill":{"duration":1.2043,"end_time":"2021-07-22T13:58:56.215653","exception":false,"start_time":"2021-07-22T13:58:55.011353","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = final_data.copy()\ndata = pd.merge(\n  data,\n  playerBoxScores[['dailyDataDate','playerId','assists', 'balls', 'baseOnBalls',\n       'baseOnBallsPitching', 'battersFaced', 'battingOrder', 'blownSaves',\n       'catchersInterference', 'catchersInterferencePitching',\n       'caughtStealing', 'caughtStealingPitching',\n       'completeGamesPitching', 'doubles', 'doublesPitching', 'earnedRuns','totalBases', 'triples', 'triplesPitching', 'wildPitches',\n       'winsPitching']],\n   on = ['dailyDataDate','playerId'],\n   how = 'left'\n   )","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:56.355853Z","iopub.status.busy":"2021-07-22T13:58:56.355065Z","iopub.status.idle":"2021-07-22T13:58:56.648342Z","shell.execute_reply":"2021-07-22T13:58:56.647696Z","shell.execute_reply.started":"2021-07-22T13:19:22.558391Z"},"papermill":{"duration":0.366592,"end_time":"2021-07-22T13:58:56.648519","exception":false,"start_time":"2021-07-22T13:58:56.281927","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:56.788306Z","iopub.status.busy":"2021-07-22T13:58:56.787618Z","iopub.status.idle":"2021-07-22T13:58:56.791909Z","shell.execute_reply":"2021-07-22T13:58:56.791388Z","shell.execute_reply.started":"2021-07-22T13:19:22.807394Z"},"papermill":{"duration":0.077332,"end_time":"2021-07-22T13:58:56.792048","exception":false,"start_time":"2021-07-22T13:58:56.714716","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.drop(['gamePk'],axis=1)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:56.928962Z","iopub.status.busy":"2021-07-22T13:58:56.928335Z","iopub.status.idle":"2021-07-22T13:58:57.030726Z","shell.execute_reply":"2021-07-22T13:58:57.031282Z","shell.execute_reply.started":"2021-07-22T13:19:22.815689Z"},"papermill":{"duration":0.172719,"end_time":"2021-07-22T13:58:57.031465","exception":false,"start_time":"2021-07-22T13:58:56.858746","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.fillna(0)\nfeature_columns = [x for x in data.columns[7:]]\ntarget_columns = [x for x in data.columns[2:6]]\ndata[feature_columns] = data[feature_columns].astype(np.float32)\n\ndata = data.fillna(0)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:57.169768Z","iopub.status.busy":"2021-07-22T13:58:57.169034Z","iopub.status.idle":"2021-07-22T13:58:57.959918Z","shell.execute_reply":"2021-07-22T13:58:57.960409Z","shell.execute_reply.started":"2021-07-22T13:19:22.891439Z"},"papermill":{"duration":0.862401,"end_time":"2021-07-22T13:58:57.960601","exception":false,"start_time":"2021-07-22T13:58:57.0982","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_outliers(data):\n    Q1 = data.quantile(0.25)\n    Q3 = data.quantile(0.75)\n    IQR = Q3 - Q1\n    data = data[~((data < (Q1 - 1.5 * IQR)) | (data > (Q3 + 1.5 * IQR))).any(axis=1)]\n    \nremove_outliers(data[feature_columns])","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:58.100082Z","iopub.status.busy":"2021-07-22T13:58:58.099464Z","iopub.status.idle":"2021-07-22T13:58:58.104574Z","shell.execute_reply":"2021-07-22T13:58:58.105083Z","shell.execute_reply.started":"2021-07-18T21:19:35.940885Z"},"papermill":{"duration":0.075897,"end_time":"2021-07-22T13:58:58.105285","exception":false,"start_time":"2021-07-22T13:58:58.029388","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def log_transform(data,feature_columns):\n    for x in feature_columns:\n        data[x] = np.log10(data[x] + 1)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:58.243925Z","iopub.status.busy":"2021-07-22T13:58:58.243186Z","iopub.status.idle":"2021-07-22T13:58:58.247327Z","shell.execute_reply":"2021-07-22T13:58:58.247823Z","shell.execute_reply.started":"2021-07-22T13:19:45.050278Z"},"papermill":{"duration":0.075193,"end_time":"2021-07-22T13:58:58.247993","exception":false,"start_time":"2021-07-22T13:58:58.1728","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split            \nx_train, x_test, y_train, y_test = train_test_split(data[feature_columns],data[target_columns],test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:58.385889Z","iopub.status.busy":"2021-07-22T13:58:58.385225Z","iopub.status.idle":"2021-07-22T13:58:58.560615Z","shell.execute_reply":"2021-07-22T13:58:58.560016Z","shell.execute_reply.started":"2021-07-22T13:19:46.277291Z"},"papermill":{"duration":0.245812,"end_time":"2021-07-22T13:58:58.560771","exception":false,"start_time":"2021-07-22T13:58:58.314959","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\nx_train = sc.fit_transform(x_train)\nx_test = sc.transform(x_test)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:58.709435Z","iopub.status.busy":"2021-07-22T13:58:58.708738Z","iopub.status.idle":"2021-07-22T13:58:58.897103Z","shell.execute_reply":"2021-07-22T13:58:58.896448Z","shell.execute_reply.started":"2021-07-22T13:19:53.325074Z"},"papermill":{"duration":0.266874,"end_time":"2021-07-22T13:58:58.897282","exception":false,"start_time":"2021-07-22T13:58:58.630408","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:59.037562Z","iopub.status.busy":"2021-07-22T13:58:59.036896Z","iopub.status.idle":"2021-07-22T13:58:59.041032Z","shell.execute_reply":"2021-07-22T13:58:59.040524Z","shell.execute_reply.started":"2021-07-22T13:19:54.575689Z"},"papermill":{"duration":0.076598,"end_time":"2021-07-22T13:58:59.041174","exception":false,"start_time":"2021-07-22T13:58:58.964576","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.iloc[:,1]","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:59.186892Z","iopub.status.busy":"2021-07-22T13:58:59.186047Z","iopub.status.idle":"2021-07-22T13:58:59.190299Z","shell.execute_reply":"2021-07-22T13:58:59.189671Z","shell.execute_reply.started":"2021-07-17T14:59:00.0831Z"},"papermill":{"duration":0.081744,"end_time":"2021-07-22T13:58:59.190441","exception":false,"start_time":"2021-07-22T13:58:59.108697","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''from sklearn.ensemble import GradientBoostingRegressor\ngb = GradientBoostingRegressor()\ngb.fit(x_train,y_train.iloc[:,1])\ngb.score(x_train,y_train.iloc[:,1])'''","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:59.333782Z","iopub.status.busy":"2021-07-22T13:58:59.332924Z","iopub.status.idle":"2021-07-22T13:58:59.336603Z","shell.execute_reply":"2021-07-22T13:58:59.337075Z","shell.execute_reply.started":"2021-07-17T15:25:34.316997Z"},"papermill":{"duration":0.078922,"end_time":"2021-07-22T13:58:59.337292","exception":false,"start_time":"2021-07-22T13:58:59.25837","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.python.keras.models import Sequential\nfrom tensorflow.python.keras.layers import Dense\nfrom keras.layers import Dense, Conv1D, Flatten\nfrom keras import backend as K\n\n#_train = x_train.reshape(x_train.shape[0],x_train.shape[1],1)\n#_test=x_test.reshape(x_test.shape[0],x_test.shape[1],1)\n\nmodel = Sequential()\nmodel.add(Dense(64,input_dim=55, kernel_initializer='he_uniform', activation='relu'))\n#odel.add(Conv1D(32,16,activation=\"relu\", input_shape=(59, 1)))\n#odel.add(Flatten())\nmodel.add(Dense(32, activation=\"relu\"))\nmodel.add(Dense(16,activation=\"relu\"))\nmodel.add(Dense(32, activation=\"relu\"))\nmodel.add(Dense(64, activation=\"relu\"))\nmodel.add(Dense(4, activation='linear'))\n\nmodel.compile(loss='mae', optimizer='sgd',metrics=['mae'])\n#K.set_value(model.optimizer.learning_rate, 0.001)\n\nhistory = model.fit(x_train, y_train, epochs=50,batch_size=100, verbose=1, validation_data=(x_test,y_test))","metadata":{"execution":{"iopub.execute_input":"2021-07-22T13:58:59.488632Z","iopub.status.busy":"2021-07-22T13:58:59.487948Z","iopub.status.idle":"2021-07-22T14:01:03.220546Z","shell.execute_reply":"2021-07-22T14:01:03.221111Z","shell.execute_reply.started":"2021-07-22T13:46:01.568263Z"},"papermill":{"duration":123.814515,"end_time":"2021-07-22T14:01:03.221345","exception":false,"start_time":"2021-07-22T13:58:59.40683","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(history.history.keys())\n# \"Loss\"\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T14:01:04.608988Z","iopub.status.busy":"2021-07-22T14:01:04.606975Z","iopub.status.idle":"2021-07-22T14:01:04.772954Z","shell.execute_reply":"2021-07-22T14:01:04.772395Z","shell.execute_reply.started":"2021-07-22T13:50:36.949293Z"},"papermill":{"duration":0.870606,"end_time":"2021-07-22T14:01:04.773192","exception":false,"start_time":"2021-07-22T14:01:03.902586","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb\n\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\n#target_columns = ['target1', 'target2', 'target3', 'target4']\n\nfor (test_df, sample_prediction_df) in iter_test:\n    \n    test_df = test_df.reset_index().rename(columns = {'index':'date'})\n    sample_prediction_df = sample_prediction_df.reset_index()\n    sample_prediction_df = sample_prediction_df.rename(columns={'date':'ddate'})\n    sample_prediction_df['date'] = pd.to_datetime(sample_prediction_df['ddate'], format='%Y%m%d')\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId'].apply(lambda x: x.split('_')[1]).astype(int)\n\n    #games\n    \n    games_nested_table = test_df[['date','games']]\n    games_nested_table = (games_nested_table.reset_index(drop = True))\n    games_test = [] \n    for date_index, date_row in games_nested_table.iterrows():\n        daily_df = pd.read_json(date_row['games'])   \n        daily_df['date'] = date_row['date']\n        games_test = games_test + [daily_df] \n        \n    games_test = (pd.concat(games_test,\n      ignore_index = True).\n      reset_index()\n      )\n    #games_test['date'] = sample_prediction_df['date']\n    games_test = games_test[['date','gamePk','isTie','gamesInSeries','seriesDescription',\n                             'homeWinPct','awayWinPct','homeId','awayId']]\n    \n    #players\n    \n    players_nested_table = test_df[['date','playerBoxScores']]\n    players_nested_table = (players_nested_table.reset_index(drop = True))\n    players_test = []    \n    for date_index, date_row in players_nested_table.iterrows():\n        daily_df = pd.read_json(date_row['playerBoxScores'])  \n        daily_df['date'] = date_row['date']\n        players_test = players_test + [daily_df]   \n    players_test = (pd.concat(players_test,\n      ignore_index = True).\n      reset_index()\n      )\n    #players_test['date'] = sample_prediction_df['date']\n    players_test = players_test[['date','playerId','gamePk','teamId','runsScored', 'atBats', 'homeRuns',\n                             'flyOuts','hits','strikes','balks','errors','chances','rbi','assists', 'balls',\n       'baseOnBalls', 'baseOnBallsPitching', 'battersFaced', 'battingOrder',\n       'blownSaves', 'catchersInterference', 'catchersInterferencePitching',\n       'caughtStealing', 'caughtStealingPitching', 'completeGamesPitching',\n       'doubles', 'doublesPitching', 'earnedRuns', 'totalBases', 'triples',\n       'triplesPitching', 'wildPitches', 'winsPitching']]\n\n    \n    #standings\n    \n    \n    standings_nested_table = test_df[['date','standings']]\n    standings_nested_table = (standings_nested_table.reset_index(drop = True))\n    standings_test = [] \n    for date_index, date_row in standings_nested_table.iterrows():\n        daily_df = pd.read_json(date_row['standings']) \n        daily_df['date'] = date_row['date']\n        standings_test = standings_test + [daily_df]    \n    standings_test = (pd.concat(standings_test,\n      ignore_index = True).\n      reset_index()\n      )\n    #standings_test['date'] = sample_prediction_df['date']\n    standings_test = standings_test[['date','teamId','wins','losses','pct','xWinLossPct','divisionRank'\n             ,'leagueRank','wildCardRank','lastTenWins','lastTenLosses']]\n    \n    \n    players_test['date'] = pd.to_datetime(players_test['date'], format='%Y%m%d')\n    final_test4 = pd.merge(\n    sample_prediction_df,\n    players_test.drop_duplicates(subset=['date','playerId']),\n    on = ['playerId','date'],\n    how = 'left'\n    )\n\n\n    games_test['date'] = pd.to_datetime(games_test['date'], format='%Y%m%d')\n    final_test3 = pd.merge(\n    final_test4,\n    games_test,\n    on = ['date','gamePk'],\n    how = 'left'\n    )\n    \n    standings_test['date'] = pd.to_datetime(standings_test['date'], format='%Y%m%d')\n    final_test2 = pd.merge(\n    final_test3,\n    standings_test,\n    on = ['date','teamId'],\n    how = 'left'\n    )\n    \n    final_test1 = pd.merge(\n    final_test2,\n    standings_test[['date','teamId','divisionRank', 'leagueRank','wildCardRank']],\n    left_on = ['date','homeId'],\n    right_on= ['date','teamId'],\n    how = 'left'\n    )\n    \n    final_test = pd.merge(\n    final_test1,\n    standings_test[['date','teamId','divisionRank', 'leagueRank','wildCardRank']],\n    left_on = ['date','awayId'],\n    right_on= ['date','teamId'],\n    how = 'left'\n    )\n    \n    final_test = final_test.rename(columns={'divisionRank_x': 'player_divisionRank', 'divisionRank_y': 'home_divisionRank', \n                                       'divisionRank' : 'away_divisionRank'})\n    final_test = final_test.rename(columns={'leagueRank_x': 'player_leagueRank', 'leagueRank_y': 'home_leagueRank', \n                                       'leagueRank' : 'away_leagueRank'})\n    final_test = final_test.rename(columns={'wildCardRank_x': 'player_wildCardRank', 'wildCardRank_y': 'home_wildCardRank', \n                                       'wildCardRank' : 'away_wildCardRank'})\n\n    final_test['divisionRank_diff'] = (final_test['home_divisionRank'] - final_test['away_divisionRank']).abs()\n    final_test['leagueRank_diff'] = (final_test['home_leagueRank'] - final_test['away_leagueRank']).abs()\n    final_test['wildCardRank_diff'] = (final_test['home_wildCardRank'] - final_test['away_wildCardRank']).abs()\n    final_test['pct_diff'] = (final_test['homeWinPct'] - final_test['awayWinPct']).abs()\n    \n    final_test = final_test.drop(['awayId','teamId','homeId','playerId','gamePk'],axis=1)\n    final_test['numberOfFollowers_x'] = 0\n    final_test['numberOfFollowers_y'] = 0\n    final_test['awardName'] = 0\n    final_test = final_test.rename(columns={'chances' : 'chances_x'})\n    final_test = final_test.fillna(0)  \n    \n    cols = ['atBats',\n 'homeRuns',\n 'flyOuts',\n 'hits',\n 'strikes',\n 'balks',\n 'errors',\n 'chances_x',\n 'rbi',\n 'wins',\n 'losses',\n 'pct',\n 'xWinLossPct',\n 'player_divisionRank',\n 'player_leagueRank',\n 'player_wildCardRank',\n 'lastTenWins',\n 'lastTenLosses',\n 'awardName',\n 'numberOfFollowers_x',\n 'numberOfFollowers_y',\n 'isTie',\n 'gamesInSeries',\n 'homeWinPct',\n 'awayWinPct',\n 'pct_diff',\n 'home_divisionRank',\n 'home_leagueRank',\n 'home_wildCardRank',\n 'away_divisionRank',\n 'away_leagueRank','away_wildCardRank', 'divisionRank_diff','leagueRank_diff','wildCardRank_diff','assists','balls',\n 'baseOnBalls', 'baseOnBallsPitching','battersFaced','battingOrder','blownSaves',\n 'catchersInterference','catchersInterferencePitching','caughtStealing','caughtStealingPitching','completeGamesPitching','doubles',\n 'doublesPitching','earnedRuns','totalBases','triples', 'triplesPitching','wildPitches','winsPitching']\n    \n    final_test = final_test[cols]\n    \n    \n    final_test = sc.transform(final_test)\n    \n    final_test = final_test.astype(np.float32)\n    sample_prediction_df = sample_prediction_df.drop(['playerId'],axis=1)\n    sample_prediction_df = sample_prediction_df.set_index('date')\n    sample_prediction_df = sample_prediction_df.rename(columns={'ddate':'date'})\n    sample_prediction_df = sample_prediction_df[['date','date_playerId','target1', 'target2', 'target3', 'target4']]\n    sample_prediction_df = sample_prediction_df.drop(['date'],axis=1)\n    sample_prediction_df['target1'] = np.clip(model.predict(final_test)[:,0], 0, 100)\n    sample_prediction_df['target2'] = np.clip(model.predict(final_test)[:,1], 0, 100)\n    sample_prediction_df['target3'] = np.clip(model.predict(final_test)[:,2], 0, 100)\n    sample_prediction_df['target4'] = np.clip(model.predict(final_test)[:,3], 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)\n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T14:01:06.174459Z","iopub.status.busy":"2021-07-22T14:01:06.141228Z","iopub.status.idle":"2021-07-22T14:01:10.96749Z","shell.execute_reply":"2021-07-22T14:01:10.966888Z","shell.execute_reply.started":"2021-07-18T16:14:54.827452Z"},"papermill":{"duration":5.509074,"end_time":"2021-07-22T14:01:10.967656","exception":false,"start_time":"2021-07-22T14:01:05.458582","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.execute_input":"2021-07-16T00:21:46.07697Z","iopub.status.busy":"2021-07-16T00:21:46.076519Z","iopub.status.idle":"2021-07-16T00:21:46.083994Z","shell.execute_reply":"2021-07-16T00:21:46.082632Z","shell.execute_reply.started":"2021-07-16T00:21:46.076925Z"},"papermill":{"duration":0.709801,"end_time":"2021-07-22T14:01:12.404907","exception":false,"start_time":"2021-07-22T14:01:11.695106","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''sample_prediction_df = sample_prediction_df.drop(['date'],axis=1)\nsample_prediction_df = sample_prediction_df.rename(columns={'ddate':'date'})\nsample_prediction_df = sample_prediction_df[['date','date_playerId','target1', 'target2', 'target3', 'target4']]\nsample_prediction_df = sample_prediction_df.drop(['date'],axis=1)'''","metadata":{"execution":{"iopub.execute_input":"2021-07-22T14:01:13.812298Z","iopub.status.busy":"2021-07-22T14:01:13.811378Z","iopub.status.idle":"2021-07-22T14:01:13.815591Z","shell.execute_reply":"2021-07-22T14:01:13.814932Z","shell.execute_reply.started":"2021-07-16T00:21:46.085847Z"},"papermill":{"duration":0.711503,"end_time":"2021-07-22T14:01:13.81585","exception":false,"start_time":"2021-07-22T14:01:13.104347","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction_df.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-22T14:01:15.248891Z","iopub.status.busy":"2021-07-22T14:01:15.247835Z","iopub.status.idle":"2021-07-22T14:01:15.254489Z","shell.execute_reply":"2021-07-22T14:01:15.253669Z","shell.execute_reply.started":"2021-07-16T00:21:46.105751Z"},"papermill":{"duration":0.733491,"end_time":"2021-07-22T14:01:15.254641","exception":false,"start_time":"2021-07-22T14:01:14.52115","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample_prediction_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.execute_input":"2021-07-22T14:01:16.689058Z","iopub.status.busy":"2021-07-22T14:01:16.688078Z","iopub.status.idle":"2021-07-22T14:01:16.691011Z","shell.execute_reply":"2021-07-22T14:01:16.691528Z","shell.execute_reply.started":"2021-07-16T00:21:46.127501Z"},"papermill":{"duration":0.698916,"end_time":"2021-07-22T14:01:16.691725","exception":false,"start_time":"2021-07-22T14:01:15.992809","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.685912,"end_time":"2021-07-22T14:01:18.073359","exception":false,"start_time":"2021-07-22T14:01:17.387447","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}