{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":" In this notebook only inference and submission task will be performed.","metadata":{"execution":{"iopub.status.busy":"2021-07-02T06:13:39.03302Z","iopub.execute_input":"2021-07-02T06:13:39.033339Z","iopub.status.idle":"2021-07-02T06:13:39.040014Z","shell.execute_reply.started":"2021-07-02T06:13:39.03331Z","shell.execute_reply":"2021-07-02T06:13:39.03856Z"}}},{"cell_type":"code","source":"import gc\nimport sys\nimport warnings\nfrom pathlib import Path\n\nimport os\n\nimport ipywidgets as widgets\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom tqdm import tqdm\n#warnings.simplefilter(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-04T07:29:54.631862Z","iopub.execute_input":"2021-07-04T07:29:54.632300Z","iopub.status.idle":"2021-07-04T07:29:55.700227Z","shell.execute_reply.started":"2021-07-04T07:29:54.632227Z","shell.execute_reply":"2021-07-04T07:29:55.699331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper function to unpack json found in daily data\ndef unpack_json(json_str):\n    return np.nan if pd.isna(json_str) else pd.read_json(json_str)\n\n# helper function to add is_played column\ndef is_played_games(row):\n    if pd.isnull(row['gameDate']):\n        is_played = 0\n    else:\n        is_played = 1\n    return is_played\n\n# helper function to add bmi column\ndef BMI(row):\n    '''\n    Calculate BMI for players.csv\n    '''\n    height_in = row['heightInches']\n    mass_lb = row['weight']\n    bmi = (mass_lb/height_in**2)*703\n    \n    return bmi","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:55.701406Z","iopub.execute_input":"2021-07-04T07:29:55.701728Z","iopub.status.idle":"2021-07-04T07:29:55.708078Z","shell.execute_reply.started":"2021-07-04T07:29:55.701700Z","shell.execute_reply":"2021-07-04T07:29:55.706519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper function to unpack json found in daily data\ndef unpack_json(json_str):\n    return np.nan if pd.isna(json_str) else pd.read_json(json_str)\n\n# helper function to add is_played column\ndef is_played_games(row):\n    if pd.isnull(row['gameDate']):\n        is_played = 0\n    else:\n        is_played = 1\n    return is_played","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:55.709830Z","iopub.execute_input":"2021-07-04T07:29:55.710131Z","iopub.status.idle":"2021-07-04T07:29:55.728130Z","shell.execute_reply.started":"2021-07-04T07:29:55.710103Z","shell.execute_reply":"2021-07-04T07:29:55.727474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def age_now(row):\n    '''\n    Calculate Age (in days) at Given Day\n    '''\n    date = row['date']\n    given_day = pd.to_datetime(date,format='%Y%m%d')\n    dob = row[\"DOB\"] #should be datetime formated already\n    age = (given_day - dob).days/365\n    \n    return age\n\ndef age_now_d(row):\n    '''\n    Calculate Age (in days) at Given Day\n    '''\n    date = row['date_playerId'].split('_')[0]\n    given_day = pd.to_datetime(date,format='%Y%m%d')\n    dob = row[\"DOB\"] #should be datetime formated already\n    age = (given_day - dob).days/365\n    \n    return age\n\ndef mlbDebutDays_now(row):\n    '''\n    Calculate mlbDebutDays at Given Day\n    '''\n    date = row['date']\n    given_day = pd.to_datetime(date,format='%Y%m%d')\n    dob = pd.to_datetime(row[\"mlbDebutDate\"])\n    mlbDebutDays = (given_day - dob).days\n    \n    return mlbDebutDays","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:55.729699Z","iopub.execute_input":"2021-07-04T07:29:55.730362Z","iopub.status.idle":"2021-07-04T07:29:55.746579Z","shell.execute_reply.started":"2021-07-04T07:29:55.730328Z","shell.execute_reply":"2021-07-04T07:29:55.745851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evalYear(row):\n    year = pd.to_datetime(row.date, format='%Y%m%d').year\n    return year","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:55.749236Z","iopub.execute_input":"2021-07-04T07:29:55.749632Z","iopub.status.idle":"2021-07-04T07:29:55.770726Z","shell.execute_reply.started":"2021-07-04T07:29:55.749581Z","shell.execute_reply":"2021-07-04T07:29:55.769368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def player_data_process(dataset):\n    '''\n    This fucntion process the players.csv\n    New Columns : age, bmi\n    '''\n    temp = dataset.copy()\n    temp[\"DOB\"] = pd.to_datetime(temp[\"DOB\"]) # death of birth\n    temp['bmi'] = temp.apply(BMI,axis=1)\n    \n    return temp","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:55.773449Z","iopub.execute_input":"2021-07-04T07:29:55.773953Z","iopub.status.idle":"2021-07-04T07:29:55.794573Z","shell.execute_reply.started":"2021-07-04T07:29:55.773916Z","shell.execute_reply":"2021-07-04T07:29:55.793501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"../input/mlb-player-digital-engagement-forecasting/example_test.csv\")\nplayers = pd.read_csv(\"../input/mlb-player-digital-engagement-forecasting/players.csv\")\nmean_target_by_player = pd.read_csv(\"../input/derived-data/mean_target_by_player.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:55.795739Z","iopub.execute_input":"2021-07-04T07:29:55.796039Z","iopub.status.idle":"2021-07-04T07:29:56.586938Z","shell.execute_reply.started":"2021-07-04T07:29:55.796006Z","shell.execute_reply":"2021-07-04T07:29:56.586162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:56.588353Z","iopub.execute_input":"2021-07-04T07:29:56.589014Z","iopub.status.idle":"2021-07-04T07:29:56.760816Z","shell.execute_reply.started":"2021-07-04T07:29:56.588976Z","shell.execute_reply":"2021-07-04T07:29:56.759330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom keras import optimizers","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:29:56.763461Z","iopub.execute_input":"2021-07-04T07:29:56.763755Z","iopub.status.idle":"2021-07-04T07:30:04.422993Z","shell.execute_reply.started":"2021-07-04T07:29:56.763731Z","shell.execute_reply":"2021-07-04T07:30:04.421686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model(n_in):\n    inp = L.Input(name=\"inputs\", shape=(n_in,))\n    x = L.Dense(50, activation=\"relu\", name=\"d3\")(inp)\n#     x = L.Dropout(0.2)(x)\n    x = L.Dense(50, activation=\"relu\", name=\"d4\")(x)\n#     x = L.Dropout(0.2)(x)\n    preds = L.Dense(4, activation=\"linear\", name=\"preds\")(x)\n    \n    model = M.Model(inp, preds, name=\"ANN\")\n    model.compile(loss=\"mean_absolute_error\", optimizer=optimizers.Adamax(lr=0.001, decay=1e-3))\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.424034Z","iopub.execute_input":"2021-07-04T07:30:04.424235Z","iopub.status.idle":"2021-07-04T07:30:04.435363Z","shell.execute_reply.started":"2021-07-04T07:30:04.424213Z","shell.execute_reply":"2021-07-04T07:30:04.433837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = make_model(7)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.437361Z","iopub.execute_input":"2021-07-04T07:30:04.437755Z","iopub.status.idle":"2021-07-04T07:30:04.565823Z","shell.execute_reply.started":"2021-07-04T07:30:04.437715Z","shell.execute_reply":"2021-07-04T07:30:04.564361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loads the weights\nmodel.load_weights(\"../input/weights/model_ANN2.cpkt\")","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.566945Z","iopub.execute_input":"2021-07-04T07:30:04.567265Z","iopub.status.idle":"2021-07-04T07:30:04.655097Z","shell.execute_reply.started":"2021-07-04T07:30:04.567232Z","shell.execute_reply":"2021-07-04T07:30:04.654379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.656032Z","iopub.execute_input":"2021-07-04T07:30:04.656409Z","iopub.status.idle":"2021-07-04T07:30:04.667355Z","shell.execute_reply.started":"2021-07-04T07:30:04.656377Z","shell.execute_reply":"2021-07-04T07:30:04.665785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction Task","metadata":{}},{"cell_type":"code","source":"FECOLS = ['t1_m','t2_m','t3_m','t4_m','is_played','age','bmi'] #feature columns \nTGTCOLS = ['target1', 'target2', 'target3', 'target4']  #target columns","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.668767Z","iopub.execute_input":"2021-07-04T07:30:04.669064Z","iopub.status.idle":"2021-07-04T07:30:04.680540Z","shell.execute_reply.started":"2021-07-04T07:30:04.669033Z","shell.execute_reply":"2021-07-04T07:30:04.679678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_date_playerid(row):\n\n    given_day = pd.to_datetime(row['date'],format='%Y%m%d') #taking timestamp of the given day\n    next_day = given_day + pd.DateOffset(1) # next date\n                                   \n    next_day = str(next_day).split(\" \")[0].replace(\"-\",\"\")\n    playerId = row['playerId']\n    date_playerId = next_day+\"_\"+str(playerId)\n\n    return date_playerId","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.681769Z","iopub.execute_input":"2021-07-04T07:30:04.682196Z","iopub.status.idle":"2021-07-04T07:30:04.706195Z","shell.execute_reply.started":"2021-07-04T07:30:04.682164Z","shell.execute_reply":"2021-07-04T07:30:04.704229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_prediction(test_df,sub_mode=True):\n    append = False #flag for append to new_df\n    \n    for i in range(test_df.shape[0]):\n        #test dataframe that is provided for submission has no formal date column\n        if sub_mode:\n            date = test_df.index[i]\n        else:\n            date = test_df.date.iloc[i] #taking the date where we are expanding json\n        \n        roster = unpack_json(test_df.rosters.iloc[i])\n        roster.insert(0,'date',date) #inserting the given date\n        \n        if append==False:\n            append= True\n            new_df = roster\n        else:\n            new_df = new_df.append(roster,ignore_index=True)\n            \n    \n    new_df['date_playerId'] = new_df.apply(add_date_playerid,axis=1)\n    return new_df","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.707919Z","iopub.execute_input":"2021-07-04T07:30:04.708245Z","iopub.status.idle":"2021-07-04T07:30:04.728917Z","shell.execute_reply.started":"2021-07-04T07:30:04.708215Z","shell.execute_reply":"2021-07-04T07:30:04.726756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#processing players data\nplayers_processed = player_data_process(players)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.730278Z","iopub.execute_input":"2021-07-04T07:30:04.730593Z","iopub.status.idle":"2021-07-04T07:30:04.798282Z","shell.execute_reply.started":"2021-07-04T07:30:04.730564Z","shell.execute_reply":"2021-07-04T07:30:04.796853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tempx = process_prediction(test,sub_mode=False)\ntempx","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:04.799431Z","iopub.execute_input":"2021-07-04T07:30:04.799750Z","iopub.status.idle":"2021-07-04T07:30:06.052485Z","shell.execute_reply.started":"2021-07-04T07:30:04.799719Z","shell.execute_reply":"2021-07-04T07:30:06.051047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tempx.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.053761Z","iopub.execute_input":"2021-07-04T07:30:06.053979Z","iopub.status.idle":"2021-07-04T07:30:06.066071Z","shell.execute_reply.started":"2021-07-04T07:30:06.053955Z","shell.execute_reply":"2021-07-04T07:30:06.064629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tempx[tempx.is_played==0]\n# tempx[tempx.playerId==596049]","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.067534Z","iopub.execute_input":"2021-07-04T07:30:06.067893Z","iopub.status.idle":"2021-07-04T07:30:06.082485Z","shell.execute_reply.started":"2021-07-04T07:30:06.067860Z","shell.execute_reply":"2021-07-04T07:30:06.080944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = tempx[FECOLS].values\n# preds = model.predict(X)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.084033Z","iopub.execute_input":"2021-07-04T07:30:06.084365Z","iopub.status.idle":"2021-07-04T07:30:06.103027Z","shell.execute_reply.started":"2021-07-04T07:30:06.084323Z","shell.execute_reply":"2021-07-04T07:30:06.102061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tempx[TGTCOLS] = np.clip(preds,0,100)\n# tempx","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.104247Z","iopub.execute_input":"2021-07-04T07:30:06.104610Z","iopub.status.idle":"2021-07-04T07:30:06.125026Z","shell.execute_reply.started":"2021-07-04T07:30:06.104576Z","shell.execute_reply":"2021-07-04T07:30:06.123382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.127361Z","iopub.execute_input":"2021-07-04T07:30:06.127786Z","iopub.status.idle":"2021-07-04T07:30:06.405798Z","shell.execute_reply.started":"2021-07-04T07:30:06.127756Z","shell.execute_reply":"2021-07-04T07:30:06.404674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"import mlb\nenv = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.407225Z","iopub.execute_input":"2021-07-04T07:30:06.407702Z","iopub.status.idle":"2021-07-04T07:30:06.445858Z","shell.execute_reply.started":"2021-07-04T07:30:06.407669Z","shell.execute_reply":"2021-07-04T07:30:06.444965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    \n    sample_prediction_df = sample_prediction_df.reset_index(drop=True)\n    sample_prediction_df.drop(TGTCOLS,axis=1,inplace=True)\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    \n    \n    # Dealing with missing values\n    if test_df['rosters'].iloc[0] == test_df['rosters'].iloc[0]:\n        test_rosters = pd.DataFrame(eval(test_df['rosters'].iloc[0]))\n    else:\n        test_rosters = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in rosters.columns:\n            if col == 'playerId': continue\n            test_rosters[col] = np.nan\n    \n    test = sample_prediction_df[['playerId']].copy()\n    test = test.merge(test_rosters, on='playerId', how='left')\n    test = test.merge(players_processed[['playerId','bmi','DOB']],on='playerId',how='left')\n    test.insert(0,'date',test_df.index[0]) #test_df fully extended here (rosters)\n\n    \n    #add new columns: age, is_played\n    test['age'] = test.apply(age_now,axis=1)\n    test['is_played'] = test.apply(is_played_games,axis=1)\n    #adding mean_target_by_player\n    test = test.merge(mean_target_by_player,on='playerId',how='left')\n    \n    \n    #making predictions : preds\n    X_ = test[FECOLS].values\n    preds = model.predict(X_)\n    \n#     #to debug\n#     sample_pred_temp = sample_pred_temp.append(sample_prediction_df)#,ignore_index=True)\n#     test_temp = test_temp.append(test) #ignore_index=True)\n#     #\n    \n    #merging prediction to submission dataframe\n    sample_prediction_df[TGTCOLS] = np.clip(preds,0,100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)    \n    del sample_prediction_df['playerId']\n   \n    env.predict(sample_prediction_df)\n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:25.021473Z","iopub.execute_input":"2021-07-04T07:30:25.021735Z","iopub.status.idle":"2021-07-04T07:30:27.391704Z","shell.execute_reply.started":"2021-07-04T07:30:25.021711Z","shell.execute_reply":"2021-07-04T07:30:27.390018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:27.393091Z","iopub.execute_input":"2021-07-04T07:30:27.393307Z","iopub.status.idle":"2021-07-04T07:30:27.405105Z","shell.execute_reply.started":"2021-07-04T07:30:27.393285Z","shell.execute_reply":"2021-07-04T07:30:27.404527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #For debugging (first uncomment 'to debug' section in the submission loop, then run twice)\n# test_temp = test.copy()\n# test_temp.drop(test_temp.index,axis=0,inplace=True)\n\n# sample_pred_temp = sample_prediction_df.copy()\n# sample_pred_temp.drop(sample_pred_temp.index,axis=0,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.694951Z","iopub.status.idle":"2021-07-04T07:30:06.695325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # to reset env.predict() to work\n# example_sample_submission = pd.read_csv(\"../input/mlb-player-digital-engagement-forecasting/example_sample_submission.csv\")\n# example_sample_submission\n# env.predict(example_sample_submission) ","metadata":{"execution":{"iopub.status.busy":"2021-07-04T07:30:06.696066Z","iopub.status.idle":"2021-07-04T07:30:06.696408Z"},"trusted":true},"execution_count":null,"outputs":[]}]}