{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nimport gc\nimport matplotlib.pyplot as plt\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-10T16:05:16.493229Z","iopub.execute_input":"2021-07-10T16:05:16.493748Z","iopub.status.idle":"2021-07-10T16:05:16.512237Z","shell.execute_reply.started":"2021-07-10T16:05:16.493631Z","shell.execute_reply":"2021-07-10T16:05:16.510787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int64)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float32)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float64)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:05:16.513558Z","iopub.execute_input":"2021-07-10T16:05:16.514105Z","iopub.status.idle":"2021-07-10T16:05:16.525333Z","shell.execute_reply.started":"2021-07-10T16:05:16.514053Z","shell.execute_reply":"2021-07-10T16:05:16.524370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplayers = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/players.csv')\n# players = reduce_mem_usage(players,verbose = False)\n\n# teams = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/teams.csv')\n# teams = reduce_mem_usage(teams,verbose = False)\n\n# seasons = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/seasons.csv')\n# seasons = reduce_mem_usage(seasons,verbose = False)\n\ntrain = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/train.csv')\n# train = reduce_mem_usage(train,verbose = False)\n\n# awards = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/awards.csv')\n# awards = reduce_mem_usage(awards,verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:05:16.526898Z","iopub.execute_input":"2021-07-10T16:05:16.527360Z","iopub.status.idle":"2021-07-10T16:06:03.483175Z","shell.execute_reply.started":"2021-07-10T16:05:16.527321Z","shell.execute_reply":"2021-07-10T16:06:03.482064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pids_test = players.playerId[players.playerForTestSetAndFuturePreds == True]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.486812Z","iopub.execute_input":"2021-07-10T16:06:03.487160Z","iopub.status.idle":"2021-07-10T16:06:03.496580Z","shell.execute_reply.started":"2021-07-10T16:06:03.487129Z","shell.execute_reply":"2021-07-10T16:06:03.495160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ## Uncomment this chunk of codes to have a basic info the columns present in each meta data and how the \n# ## first few rows look\n# display(players.info())\n# display(teams.head())\n# display(teams.info())\n# display(seasons.head())\n# display(seasons.info())\n# display(train.head())\n# display(train.info())\n# display(awards.head())\n# display(awards.info())","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.498544Z","iopub.execute_input":"2021-07-10T16:06:03.499070Z","iopub.status.idle":"2021-07-10T16:06:03.511047Z","shell.execute_reply.started":"2021-07-10T16:06:03.499021Z","shell.execute_reply":"2021-07-10T16:06:03.509695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%who","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.513156Z","iopub.execute_input":"2021-07-10T16:06:03.514255Z","iopub.status.idle":"2021-07-10T16:06:03.529014Z","shell.execute_reply.started":"2021-07-10T16:06:03.514197Z","shell.execute_reply":"2021-07-10T16:06:03.527891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unpack_raw_data(raw_data,dfs_name):\n    \n    unnested_data_dict = dict()\n# columns = train.drop('date', axis = 1).columns.values.tolist()\n# columns\n\n    for col in dfs_name:\n\n        data_nested_info = raw_data[['date',col]]\n\n        data_nested_info = (data_nested_info[\n              ~pd.isna(data_nested_info[col])\n              ].\n              reset_index(drop = True)\n              )\n\n        daily_dfs_collection = []\n        for data_index, data_row in data_nested_info.iterrows():\n            daily_df = pd.read_json(data_row[col])\n\n            daily_df['dailydate'] = data_row['date']\n\n            daily_dfs_collection = daily_dfs_collection + [daily_df]\n\n        unnested_table = (pd.concat(daily_dfs_collection,\n              ignore_index = True).\n                # Set and reset index to move 'dailyDataDate' to front of df\n              set_index('dailydate').\n              reset_index()\n              )\n\n    #     display(col)\n        unnested_table = reduce_mem_usage(unnested_table,verbose = False)\n\n        unnested_data_dict[col] = unnested_table\n        del daily_dfs_collection,unnested_table\n    return unnested_data_dict \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.530173Z","iopub.execute_input":"2021-07-10T16:06:03.530453Z","iopub.status.idle":"2021-07-10T16:06:03.541477Z","shell.execute_reply.started":"2021-07-10T16:06:03.530424Z","shell.execute_reply":"2021-07-10T16:06:03.540153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check = unpack_raw_data(train,['playerBoxScores','games'])\nfeatures = ['dailydate','engagementMetricsDate','target1','target2','target3','target4','flyOuts','strikeOuts','stolenBases','homeRunsPitching']\n   ","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.544603Z","iopub.execute_input":"2021-07-10T16:06:03.545095Z","iopub.status.idle":"2021-07-10T16:06:03.560640Z","shell.execute_reply.started":"2021-07-10T16:06:03.545032Z","shell.execute_reply":"2021-07-10T16:06:03.559173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_train_data(raw_data,features):\n    \n    nday_pl_eng = raw_data['nextDayPlayerEngagement']\n    pl_box_scores = raw_data['playerBoxScores']\n  \n    pl_eng_w_scores = pd.merge(nday_pl_eng,pl_box_scores,on=['dailydate','playerId'],how = 'inner')\n    train_data = pl_eng_w_scores[features]\n    del nday_pl_eng,pl_box_scores\n    return train_data,features\n","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.562850Z","iopub.execute_input":"2021-07-10T16:06:03.563178Z","iopub.status.idle":"2021-07-10T16:06:03.575107Z","shell.execute_reply.started":"2021-07-10T16:06:03.563138Z","shell.execute_reply":"2021-07-10T16:06:03.573695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nraw_data = unpack_raw_data(train,['nextDayPlayerEngagement','playerBoxScores'])\ntrain_data,features = make_train_data(raw_data,features)\ndel(raw_data)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:03.576468Z","iopub.execute_input":"2021-07-10T16:06:03.576787Z","iopub.status.idle":"2021-07-10T16:06:50.554697Z","shell.execute_reply.started":"2021-07-10T16:06:03.576734Z","shell.execute_reply":"2021-07-10T16:06:50.552670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:50.556348Z","iopub.execute_input":"2021-07-10T16:06:50.556676Z","iopub.status.idle":"2021-07-10T16:06:50.660965Z","shell.execute_reply.started":"2021-07-10T16:06:50.556646Z","shell.execute_reply":"2021-07-10T16:06:50.659461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.fillna(-1,inplace = True)\nsample_y = train_data[['target1','target2','target3','target4']]\nsample_X = train_data[['flyOuts','strikeOuts','stolenBases','homeRunsPitching']]\n\n# display(sample_X.head())\n# display(sample_y.head())\nfrom sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test = train_test_split(sample_X,sample_y)\ndel train_data,sample_X,sample_y","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:50.662939Z","iopub.execute_input":"2021-07-10T16:06:50.663290Z","iopub.status.idle":"2021-07-10T16:06:51.110819Z","shell.execute_reply.started":"2021-07-10T16:06:50.663258Z","shell.execute_reply":"2021-07-10T16:06:51.109501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%who","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:51.112513Z","iopub.execute_input":"2021-07-10T16:06:51.112872Z","iopub.status.idle":"2021-07-10T16:06:51.123649Z","shell.execute_reply.started":"2021-07-10T16:06:51.112839Z","shell.execute_reply":"2021-07-10T16:06:51.121556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense\nmodel = Sequential()\nmodel.add(Dense(6,input_dim = 4,activation = 'relu'))\nmodel.add(Dense(6,activation = 'relu'))\nmodel.add(Dense(4))\nmodel.compile(optimizer = 'adam',loss = 'mae', metrics = ['mae'])\n\nfit_model = model.fit(X_train,y_train,validation_data=(X_test,y_test),epochs=5)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:06:51.126057Z","iopub.execute_input":"2021-07-10T16:06:51.126500Z","iopub.status.idle":"2021-07-10T16:07:22.913876Z","shell.execute_reply.started":"2021-07-10T16:06:51.126464Z","shell.execute_reply":"2021-07-10T16:07:22.912961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['flyOuts','strikeOuts','stolenBases','homeRunsPitching']\nprimary_cols = 'playerBoxScores'","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:07:22.915215Z","iopub.execute_input":"2021-07-10T16:07:22.915655Z","iopub.status.idle":"2021-07-10T16:07:22.919549Z","shell.execute_reply.started":"2021-07-10T16:07:22.915623Z","shell.execute_reply":"2021-07-10T16:07:22.918740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_features(unnested_data_dict,primary_cols,features,sample_prediction_df):\n    \n    test_set = unnested_data_dict[primary_cols]\n    tmp = features.copy()\n    tmp.append('playerId')\n    test_set = test_set[tmp]\n    test_set = test_set.groupby('playerId').sum().reset_index()\n    test_set = test_set.merge(pids_test,on = 'playerId',how = 'right')\n    test_set = test_set.fillna(-1)\n    sub_df = sample_prediction_df.copy()\n    sub_df['playerId'] = sub_df['date_playerId'].map(lambda x: int(x.split('_')[1]))\n    test_set = sub_df.merge(test_set,on = 'playerId', how = 'left')\n    test_set = test_set[features]\n\n    return test_set ","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:07:22.920579Z","iopub.execute_input":"2021-07-10T16:07:22.920911Z","iopub.status.idle":"2021-07-10T16:07:22.936201Z","shell.execute_reply.started":"2021-07-10T16:07:22.920878Z","shell.execute_reply":"2021-07-10T16:07:22.935327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:07:22.937597Z","iopub.execute_input":"2021-07-10T16:07:22.938197Z","iopub.status.idle":"2021-07-10T16:07:23.141949Z","shell.execute_reply.started":"2021-07-10T16:07:22.938147Z","shell.execute_reply":"2021-07-10T16:07:23.141081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in iter_test:\n    \n    test_df = test_df.reset_index().rename(columns = {'index':'date'})\n    test_data = unpack_raw_data(test_df,['playerBoxScores'])\n    \n    fit_data = make_features(test_data,primary_cols,features,sample_prediction_df)\n    \n    pred = model.predict(fit_data)\n    pred = pred.clip(0,100)\n    pred = pred.round(2)\n    \n    sample_prediction_df[['target1','target2','target3','target4']] = pred\n                        \n    \n    \n#     sample_prediction_df['target1'] = 0.4\n#     sample_prediction_df['target2'] = 2\n#     sample_prediction_df['target3'] = 0.4\n#     sample_prediction_df['target4'] = 0.7\n    \n#     sample_prediction_df = sample_prediction_df[['date_playerId']].reset_index().merge(submission,\n#                                 how='left', on='date_playerId').set_index('date')\n#     del submission\n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T16:07:23.143362Z","iopub.execute_input":"2021-07-10T16:07:23.144037Z","iopub.status.idle":"2021-07-10T16:07:24.838794Z","shell.execute_reply.started":"2021-07-10T16:07:23.143939Z","shell.execute_reply":"2021-07-10T16:07:24.837549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}