{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello. This is my first competition.\n\nCredit to @ulrich07 and @mlconsult - I simply took their notebook and filtered out all but 2021 May and Apr player targets and then medianed for each player.\n\nHere is their noteboook that I forked.\n\nhttps://www.kaggle.com/ulrich07/baseline-model-player-mean-or-median\nhttps://www.kaggle.com/mlconsult/baseline-average-1-47","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport gc\nfrom tqdm import tqdm\n\nimport mlb\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-15T05:45:48.932696Z","iopub.execute_input":"2021-06-15T05:45:48.933053Z","iopub.status.idle":"2021-06-15T05:45:48.957006Z","shell.execute_reply.started":"2021-06-15T05:45:48.933025Z","shell.execute_reply":"2021-06-15T05:45:48.955533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = \"../input/mlb-player-digital-engagement-forecasting\"","metadata":{"execution":{"iopub.status.busy":"2021-06-15T06:29:01.022788Z","iopub.execute_input":"2021-06-15T06:29:01.023152Z","iopub.status.idle":"2021-06-15T06:29:01.027404Z","shell.execute_reply.started":"2021-06-15T06:29:01.023119Z","shell.execute_reply":"2021-06-15T06:29:01.026145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read train.csv\ntr = pd.read_csv(f\"{ROOT_DIR}/train.csv\")\nprint(tr.shape)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-15T06:29:01.779655Z","iopub.execute_input":"2021-06-15T06:29:01.780046Z","iopub.status.idle":"2021-06-15T06:29:57.050521Z","shell.execute_reply.started":"2021-06-15T06:29:01.780017Z","shell.execute_reply":"2021-06-15T06:29:57.047467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-15T06:29:57.055126Z","iopub.execute_input":"2021-06-15T06:29:57.055487Z","iopub.status.idle":"2021-06-15T06:29:57.091634Z","shell.execute_reply.started":"2021-06-15T06:29:57.055456Z","shell.execute_reply":"2021-06-15T06:29:57.090875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## COMPUTING PLAYER MEAN","metadata":{}},{"cell_type":"code","source":"# Create list from nextDayPlayerEngagement column\nN_DATES = tr.shape[0]\nd = []\nfor idx in tqdm(range(N_DATES)):\n    u = eval(tr.iloc[idx, 1])\n    d += u\n#================","metadata":{"execution":{"iopub.status.busy":"2021-06-15T06:29:57.093017Z","iopub.execute_input":"2021-06-15T06:29:57.093434Z","iopub.status.idle":"2021-06-15T06:31:08.201559Z","shell.execute_reply.started":"2021-06-15T06:29:57.093393Z","shell.execute_reply":"2021-06-15T06:31:08.200474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create DataFrame from list\ntgt_df = pd.DataFrame(d)\nprint(tgt_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-15T06:31:08.203592Z","iopub.execute_input":"2021-06-15T06:31:08.204005Z","iopub.status.idle":"2021-06-15T06:31:13.798556Z","shell.execute_reply.started":"2021-06-15T06:31:08.203958Z","shell.execute_reply":"2021-06-15T06:31:13.797257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add year column\ntgt_df['year'] = pd.DatetimeIndex(tgt_df['engagementMetricsDate']).year\ntgt_df['month'] = pd.DatetimeIndex(tgt_df['engagementMetricsDate']).month","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:33:11.849853Z","iopub.execute_input":"2021-06-15T07:33:11.85021Z","iopub.status.idle":"2021-06-15T07:33:13.684355Z","shell.execute_reply.started":"2021-06-15T07:33:11.850177Z","shell.execute_reply":"2021-06-15T07:33:13.683098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tgt_df","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:33:15.920046Z","iopub.execute_input":"2021-06-15T07:33:15.920443Z","iopub.status.idle":"2021-06-15T07:33:15.941545Z","shell.execute_reply.started":"2021-06-15T07:33:15.920392Z","shell.execute_reply":"2021-06-15T07:33:15.940736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select only 2021\nnew_df = tgt_df[tgt_df['year'] == 2021]\nnew_df = new_df[new_df['month'] >= 4]\nnew_df","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:35:02.331174Z","iopub.execute_input":"2021-06-15T07:35:02.331606Z","iopub.status.idle":"2021-06-15T07:35:02.368577Z","shell.execute_reply.started":"2021-06-15T07:35:02.331575Z","shell.execute_reply":"2021-06-15T07:35:02.367622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### group by playerId setting median\nplayer_mean = new_df.groupby([\"playerId\"])[[\"target1\",\"target2\",\"target3\",\"target4\"]].median().reset_index()\ngc.collect()\nprint(player_mean.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:35:26.175271Z","iopub.execute_input":"2021-06-15T07:35:26.175645Z","iopub.status.idle":"2021-06-15T07:35:26.541756Z","shell.execute_reply.started":"2021-06-15T07:35:26.175613Z","shell.execute_reply":"2021-06-15T07:35:26.540669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_mean.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:35:32.684241Z","iopub.execute_input":"2021-06-15T07:35:32.684632Z","iopub.status.idle":"2021-06-15T07:35:32.697476Z","shell.execute_reply.started":"2021-06-15T07:35:32.6846Z","shell.execute_reply":"2021-06-15T07:35:32.696326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_pred(df):\n    df[\"playerId\"] = df[\"date_playerId\"].apply(lambda x: int( x.split(\"_\")[1] ) )\n    df.drop([\"target1\",\"target2\",\"target3\",\"target4\"], axis=1, inplace=True)\n    df = df.merge(player_mean, on=\"playerId\", how=\"left\")\n    df.drop(\"playerId\", axis=1, inplace=True)\n    df = df.fillna(0.)\n    return df\n#===================","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:35:37.26122Z","iopub.execute_input":"2021-06-15T07:35:37.2616Z","iopub.status.idle":"2021-06-15T07:35:37.267688Z","shell.execute_reply.started":"2021-06-15T07:35:37.26157Z","shell.execute_reply":"2021-06-15T07:35:37.266491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    sample_prediction_df = process_pred(sample_prediction_df)  # Create prediction\n    env.predict(sample_prediction_df)                          # Submit","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:35:39.697164Z","iopub.execute_input":"2021-06-15T07:35:39.697629Z","iopub.status.idle":"2021-06-15T07:35:39.702377Z","shell.execute_reply.started":"2021-06-15T07:35:39.697593Z","shell.execute_reply":"2021-06-15T07:35:39.701129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-15T07:35:41.259185Z","iopub.execute_input":"2021-06-15T07:35:41.259589Z","iopub.status.idle":"2021-06-15T07:35:41.273592Z","shell.execute_reply.started":"2021-06-15T07:35:41.259557Z","shell.execute_reply":"2021-06-15T07:35:41.272488Z"},"trusted":true},"execution_count":null,"outputs":[]}]}