{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport gc\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-12T10:35:19.474521Z","iopub.execute_input":"2021-06-12T10:35:19.474863Z","iopub.status.idle":"2021-06-12T10:35:19.479271Z","shell.execute_reply.started":"2021-06-12T10:35:19.474834Z","shell.execute_reply":"2021-06-12T10:35:19.478271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Config","metadata":{}},{"cell_type":"code","source":"### DATA CONFIG\n\ntargets = ['target1','target2','target3','target4']\nSPLIT = pd.to_datetime('2020-01-01')\nfeatures = ['have_game']\nidentifiers = ['playerId','date']\n\n### MODEL CONFIG\n\n","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:35:20.113406Z","iopub.execute_input":"2021-06-12T10:35:20.113968Z","iopub.status.idle":"2021-06-12T10:35:20.119511Z","shell.execute_reply.started":"2021-06-12T10:35:20.113916Z","shell.execute_reply":"2021-06-12T10:35:20.118776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_metric(ground_truth,predicted):\n    ground_truth_sorted = ground_truth.sort_values(identifiers).reset_index(drop=True)\n    predicted_sorted = predicted.sort_values(identifiers).reset_index(drop=True)\n    metric = (ground_truth_sorted[targets]-predicted_sorted[targets]).abs().mean()\n    metric.loc['CV'] = metric.mean()\n    return metric\n\ndef pair_correlation(df1,df2):\n    correlation_dfs = pd.merge(df1,df2,on=identifiers).corr()\n    cols1 = [x for x in df1.columns if x in correlation_dfs.columns and x not in df2.columns]\n    cols2 = [x for x in df2.columns if x in correlation_dfs.columns and x not in df1.columns]\n    return correlation_dfs.loc[cols1,cols2]","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:45:15.639087Z","iopub.execute_input":"2021-06-12T10:45:15.639443Z","iopub.status.idle":"2021-06-12T10:45:15.648085Z","shell.execute_reply.started":"2021-06-12T10:45:15.639414Z","shell.execute_reply":"2021-06-12T10:45:15.646955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read Data","metadata":{}},{"cell_type":"code","source":"engagements = pd.read_csv('../input/mlb-train-processed-data/nextDayPlayerEngagement.csv',index_col=0).rename({'engagementMetricsDate':'date'},axis=1)\nengagements['date'] = pd.to_datetime(engagements['date'])-pd.to_timedelta('1 days')\nplayer_box_scores = pd.read_csv('../input/mlb-train-processed-data/playerBoxScores.csv',index_col=0).rename(columns={'gameDate':'date'})\nplayer_box_scores['date'] = pd.to_datetime(player_box_scores['date'])\nplayer_box_scores['have_game'] = 1\nplayer_box_scores = player_box_scores.drop_duplicates(['playerId','date'],keep='first')\ndf = pd.merge(engagements,player_box_scores,on=['playerId','date'],how='left')\ndf['have_game'] = df['have_game'].fillna(0)\ndel player_box_scores,engagements\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:45:16.261973Z","iopub.execute_input":"2021-06-12T10:45:16.262371Z","iopub.status.idle":"2021-06-12T10:45:24.188358Z","shell.execute_reply.started":"2021-06-12T10:45:16.262337Z","shell.execute_reply":"2021-06-12T10:45:24.187421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('have_game')[targets].describe().T","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:45:24.190337Z","iopub.execute_input":"2021-06-12T10:45:24.190771Z","iopub.status.idle":"2021-06-12T10:45:26.267775Z","shell.execute_reply.started":"2021-06-12T10:45:24.190725Z","shell.execute_reply":"2021-06-12T10:45:26.266764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlations = df.corr()[['target1','target2','target3','target4']]\ncorrelations['mean_corr'] = correlations.mean(axis=1)\ncorrelations.sort_values('mean_corr',ascending=False).head(30)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:45:26.269501Z","iopub.execute_input":"2021-06-12T10:45:26.269789Z","iopub.status.idle":"2021-06-12T10:45:41.756029Z","shell.execute_reply.started":"2021-06-12T10:45:26.269762Z","shell.execute_reply":"2021-06-12T10:45:41.754959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.have_game==1].sample(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:45:41.757449Z","iopub.execute_input":"2021-06-12T10:45:41.757742Z","iopub.status.idle":"2021-06-12T10:45:41.883690Z","shell.execute_reply.started":"2021-06-12T10:45:41.757714Z","shell.execute_reply":"2021-06-12T10:45:41.882675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.have_game==0].sample(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:45:41.884869Z","iopub.execute_input":"2021-06-12T10:45:41.885172Z","iopub.status.idle":"2021-06-12T10:45:42.852504Z","shell.execute_reply.started":"2021-06-12T10:45:41.885143Z","shell.execute_reply":"2021-06-12T10:45:42.851470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Truncated Validation","metadata":{}},{"cell_type":"code","source":"## Train Targets\ntrain_targets = df.loc[df.date<SPLIT,identifiers+targets].reset_index(drop=True)\nval_targets = df.loc[df.date>=SPLIT,identifiers+targets].reset_index(drop=True)\nprint(train_targets.shape,val_targets.shape)\n\n## Train Features\ntrain_features = df.loc[df.date<SPLIT,identifiers+features].reset_index(drop=True)\nval_features = df.loc[df.date>=SPLIT,identifiers+features].reset_index(drop=True)\nprint(train_features.shape,val_features.shape)\n\n## Compute Aggregate Features From Train\naggregate = train_targets[train_features.have_game==0].groupby('playerId')[targets].median().reset_index()\naggregate.columns = ['agg_'+x if 'target' in x else x for x in aggregate.columns]\n\ntrain_features = pd.merge(train_features,aggregate,on='playerId')\nval_features = pd.merge(val_features,aggregate,on='playerId')\nprint(train_features.shape,val_features.shape)\nprint(train_features.date.min(),train_features.date.max(),val_features.date.min(),val_features.date.max())\ntrain_features.sample(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:03.640590Z","iopub.execute_input":"2021-06-12T11:22:03.640966Z","iopub.status.idle":"2021-06-12T11:22:06.127834Z","shell.execute_reply.started":"2021-06-12T11:22:03.640933Z","shell.execute_reply":"2021-06-12T11:22:06.126755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pair_correlation(train_targets,train_features)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:06.129570Z","iopub.execute_input":"2021-06-12T11:22:06.129985Z","iopub.status.idle":"2021-06-12T11:22:07.294509Z","shell.execute_reply.started":"2021-06-12T11:22:06.129944Z","shell.execute_reply":"2021-06-12T11:22:07.293569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pair_correlation(val_targets,val_features)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:07.296420Z","iopub.execute_input":"2021-06-12T11:22:07.296723Z","iopub.status.idle":"2021-06-12T11:22:08.049269Z","shell.execute_reply.started":"2021-06-12T11:22:07.296695Z","shell.execute_reply":"2021-06-12T11:22:08.047910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Modelling","metadata":{}},{"cell_type":"code","source":"selected_features = ['have_game','agg_target1', 'agg_target2', 'agg_target3', 'agg_target4']","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:08.257894Z","iopub.execute_input":"2021-06-12T11:22:08.258297Z","iopub.status.idle":"2021-06-12T11:22:08.262884Z","shell.execute_reply.started":"2021-06-12T11:22:08.258264Z","shell.execute_reply":"2021-06-12T11:22:08.261667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Regressor():\n\n    def fit(self,X,y):\n        temp = pd.concat([X,y],axis=1)\n        temp.columns = ['have_game','_','target']\n        neg,pos = temp.groupby('have_game').target.median().values\n        self.offset = (pos - neg)\n        \n    def predict(self,X):\n        offset = X.values[:,0]*self.offset\n        return np.clip(X.values[:,1] + offset,0,100)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:39.456481Z","iopub.execute_input":"2021-06-12T11:22:39.456861Z","iopub.status.idle":"2021-06-12T11:22:39.463231Z","shell.execute_reply.started":"2021-06-12T11:22:39.456826Z","shell.execute_reply":"2021-06-12T11:22:39.462167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import Ridge\n# reg = RandomForestRegressor(n_estimators=100,max_depth=6, random_state=0,n_jobs=-1,verbose=1)\npredicted = train_targets.sort_values(identifiers).reset_index(drop=True)[identifiers]\ngt = train_targets.sort_values(identifiers).reset_index(drop=True)\nregs = {}\nfor target in targets:\n    selected_features = features + ['agg_'+target]\n#     reg = Ridge(alpha=1.0)\n    reg = Regressor()\n    X = train_features.sort_values(identifiers).reset_index(drop=True)[selected_features]\n    y = train_targets.sort_values(identifiers).reset_index(drop=True)[target]\n    reg.fit(X, y)\n    print(target,reg.offset)\n    predicted[target] = reg.predict(X)\n    regs[target] = reg\n# predicted = pd.DataFrame(predicted,columns=targets)\nprint(\"Train Metrics: \",compute_metric(gt,predicted).to_dict())","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:39.933961Z","iopub.execute_input":"2021-06-12T11:22:39.934369Z","iopub.status.idle":"2021-06-12T11:22:45.799064Z","shell.execute_reply.started":"2021-06-12T11:22:39.934333Z","shell.execute_reply":"2021-06-12T11:22:45.797975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.merge(predicted,train_features,on=identifiers).groupby('have_game')[targets].mean()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:45.802181Z","iopub.execute_input":"2021-06-12T11:22:45.802467Z","iopub.status.idle":"2021-06-12T11:22:46.283399Z","shell.execute_reply.started":"2021-06-12T11:22:45.802439Z","shell.execute_reply":"2021-06-12T11:22:46.282155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.merge(train_targets,train_features,on=identifiers).groupby('have_game')[targets].mean()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:46.285803Z","iopub.execute_input":"2021-06-12T11:22:46.286279Z","iopub.status.idle":"2021-06-12T11:22:46.982011Z","shell.execute_reply.started":"2021-06-12T11:22:46.286232Z","shell.execute_reply":"2021-06-12T11:22:46.980926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = val_targets.sort_values(identifiers).reset_index(drop=True)[identifiers]\nfor target in tqdm(targets):\n    selected_features = features + ['agg_'+target]\n    X = val_features.sort_values(identifiers).reset_index(drop=True)[selected_features]\n    predicted[target] = regs[target].predict(X)\nprint(val_targets.shape,predicted.shape)\nprint(\"Val Metrics: \",compute_metric(val_targets,predicted).to_dict())","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:22:46.983894Z","iopub.execute_input":"2021-06-12T11:22:46.984201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.merge(predicted,val_features,on=identifiers).groupby('have_game')[targets].mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.merge(val_targets,val_features,on=identifiers).groupby('have_game')[targets].mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([train_targets.groupby('playerId').median().mean(),predicted.mean(),val_targets.groupby('playerId').median().mean()],axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([train_targets.mean(),predicted.mean(),val_targets.mean()],axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:21.364697Z","iopub.execute_input":"2021-06-12T11:16:21.365086Z","iopub.status.idle":"2021-06-12T11:16:21.463916Z","shell.execute_reply.started":"2021-06-12T11:16:21.365035Z","shell.execute_reply":"2021-06-12T11:16:21.462659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training For Test","metadata":{}},{"cell_type":"code","source":"%%time\n\n## Train Targets & Features\ntrain_targets = df[identifiers+targets].reset_index(drop=True)\ntrain_features = df[identifiers+features].reset_index(drop=True)\n\n## Compute Aggregate Features From Train\naggregate = train_targets.groupby('playerId')[targets].median().reset_index()\naggregate.columns = ['agg_'+x if 'target' in x else x for x in aggregate.columns]\ntrain_features = pd.merge(train_features,aggregate,on='playerId')\n\nprint(train_features.shape,train_targets.shape)\ntrain_features.sample(3)\n\npair_correlation(train_targets,train_features)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:21:18.285851Z","iopub.execute_input":"2021-06-12T10:21:18.286242Z","iopub.status.idle":"2021-06-12T10:21:21.637456Z","shell.execute_reply.started":"2021-06-12T10:21:18.286208Z","shell.execute_reply":"2021-06-12T10:21:21.636035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npredicted = train_targets.sort_values(identifiers).reset_index(drop=True)[identifiers]\ngt = train_targets.sort_values(identifiers).reset_index(drop=True)\nregs = {}\nfor target in tqdm(targets):\n    selected_features = features + ['agg_'+target]\n    X = train_features.sort_values(identifiers).reset_index(drop=True)[selected_features]\n    y = train_targets.sort_values(identifiers).reset_index(drop=True)[target]\n    reg.fit(X, y)\n    predicted[target] = reg.predict(X)\n    regs[target] = reg\nprint(\"Train Metrics: \",compute_metric(gt,predicted).to_dict())","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:21:21.641303Z","iopub.execute_input":"2021-06-12T10:21:21.641640Z","iopub.status.idle":"2021-06-12T10:21:33.509799Z","shell.execute_reply.started":"2021-06-12T10:21:21.641611Z","shell.execute_reply":"2021-06-12T10:21:33.508513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submitting","metadata":{}},{"cell_type":"code","source":"import json\nimport matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:21:33.511834Z","iopub.execute_input":"2021-06-12T10:21:33.512203Z","iopub.status.idle":"2021-06-12T10:21:33.521674Z","shell.execute_reply.started":"2021-06-12T10:21:33.512169Z","shell.execute_reply":"2021-06-12T10:21:33.520016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb\nfrom tqdm.auto import tqdm\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set\n\nfor (test_df, sample_prediction_df) in tqdm(iter_test):\n    template = sample_prediction_df[['date_playerId']].reset_index()\n    template['playerId'] = template.date_playerId.apply(lambda x:x.split('_')[1]).astype(int)\n    test_box_scores = test_df['playerBoxScores'].fillna('[]').apply(lambda x:json.loads(x))\n    test_box_scores = list(np.concatenate(test_box_scores.values))\n    test_box_scores = pd.DataFrame(test_box_scores).rename(columns={'gameDate':'date'})\n    test_box_scores['date'] = test_box_scores.date.apply(lambda x:x.replace('-','')).astype(int)\n    test_box_scores['have_game'] = 1\n    test_box_scores = test_box_scores.drop_duplicates(identifiers)\n    df = pd.merge(template,test_box_scores,on=identifiers,how='left')\n    df['have_game'] = df['have_game'].fillna(0)\n    test_features = df[identifiers+features+['date_playerId']].reset_index(drop=True)\n    test_features = pd.merge(test_features,aggregate,on='playerId').sort_values(identifiers).reset_index(drop=True)\n    predicted = pd.DataFrame(index=df.date)\n    for target in tqdm(targets):\n        selected_features = features + ['agg_'+target]\n        X = test_features[selected_features]\n        predicted[target] = regs[target].predict(X)\n\n    predicted['date_playerId'] = test_features['date_playerId'].values\n    predicted.index = df.date\n    print(predicted.shape,sample_prediction_df.shape)\n    assert predicted.shape==sample_prediction_df.shape\n    env.predict(predicted)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:21:42.183789Z","iopub.execute_input":"2021-06-12T10:21:42.184188Z","iopub.status.idle":"2021-06-12T10:21:44.329102Z","shell.execute_reply.started":"2021-06-12T10:21:42.184157Z","shell.execute_reply":"2021-06-12T10:21:44.327943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}