{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Credit to @columbia2131 - I started with his notebook and then added an external data set with descriptive statistics of the targets for each player.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## About Dataset","metadata":{}},{"cell_type":"markdown","source":"Train.csv is stored as a csv file with each column as follows.  \ntrain.csvを以下のようにして各カラムをcsvファイルとして保管しています。\n\nTo use many data, I used fruction of \"reduce_mem_usage\" to reduce CPU load.\nCPU負荷を抑えるためにreduce_mem_usageという関数を使っています。\n\nParams are tuned by Light GBM tuner. \nパラメータはLight GBM tunerで調整しています。\n\nI want to continue feature engineering, because there are other features not used.\n特徴量エンジニアリングを続けたい、まだ使っていない特徴量があるため。","metadata":{}},{"cell_type":"code","source":"%%capture\n\"\"\"\n!pip install pandarallel \n\nimport gc\n\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\nfrom pandarallel import pandarallel\npandarallel.initialize()\n\nBASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\ntrain = pd.read_csv(BASE_DIR / 'train.csv')\n\nnull = np.nan\ntrue = True\nfalse = False\n\nfor col in train.columns:\n\n    if col == 'date': continue\n\n    _index = train[col].notnull()\n    train.loc[_index, col] = train.loc[_index, col].parallel_apply(lambda x: eval(x))\n\n    outputs = []\n    for index, date, record in train.loc[_index, ['date', col]].itertuples():\n        _df = pd.DataFrame(record)\n        _df['index'] = index\n        _df['date'] = date\n        outputs.append(_df)\n\n    outputs = pd.concat(outputs).reset_index(drop=True)\n\n    outputs.to_csv(f'{col}_train.csv', index=False)\n    outputs.to_pickle(f'{col}_train.pkl')\n\n    del outputs\n    del train[col]\n    gc.collect()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:27.621537Z","iopub.execute_input":"2021-07-31T04:08:27.621982Z","iopub.status.idle":"2021-07-31T04:08:27.633008Z","shell.execute_reply.started":"2021-07-31T04:08:27.621946Z","shell.execute_reply":"2021-07-31T04:08:27.631534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{"execution":{"iopub.status.busy":"2021-06-16T09:14:33.869464Z","iopub.execute_input":"2021-06-16T09:14:33.869905Z","iopub.status.idle":"2021-06-16T09:14:33.874766Z","shell.execute_reply.started":"2021-06-16T09:14:33.869879Z","shell.execute_reply":"2021-06-16T09:14:33.873097Z"}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.metrics import mean_absolute_error\nfrom datetime import timedelta\nfrom functools import reduce\nfrom tqdm import tqdm\nimport lightgbm as lgbm\nimport mlb\nimport gc\n\npd.options.display.max_rows = 200\npd.options.display.max_columns = 100","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:29.484475Z","iopub.execute_input":"2021-07-31T04:08:29.484885Z","iopub.status.idle":"2021-07-31T04:08:31.827813Z","shell.execute_reply.started":"2021-07-31T04:08:29.484851Z","shell.execute_reply":"2021-07-31T04:08:31.826857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fruction to reduce CPU load","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:31.829365Z","iopub.execute_input":"2021-07-31T04:08:31.829663Z","iopub.status.idle":"2021-07-31T04:08:31.844363Z","shell.execute_reply.started":"2021-07-31T04:08:31.829634Z","shell.execute_reply":"2021-07-31T04:08:31.843084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = Path('../input/mlb-player-digital-engagement-forecasting')\nTRAIN_DIR = Path('../input/mlb-pdef-train-dataset')","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:32.094884Z","iopub.execute_input":"2021-07-31T04:08:32.095304Z","iopub.status.idle":"2021-07-31T04:08:32.099219Z","shell.execute_reply.started":"2021-07-31T04:08:32.095268Z","shell.execute_reply":"2021-07-31T04:08:32.098193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Select columns","metadata":{}},{"cell_type":"code","source":"targets_cols = [\n    'playerId', \n    'target1', \n    'target2', \n    'target3', \n    'target4', \n    'date'\n]\n\nplayers_cols = [\n    'playerId', \n    'primaryPositionName'\n]\n\nteams_cols = [\n    'id', \n#     'name', \n#     'teamName', \n#     'teamCode', \n#     'shortName', \n#     'abbreviation', \n#     'locationName', \n    'leagueId', \n#     'leagueName', \n    'divisionId', \n#     'divisionName', \n#     'venueId', \n#     'venueName'\n]\n\nrosters_cols = [\n    'playerId', \n    'teamId', \n    'status', \n    'date'\n]\n\nscores_cols = [\n    'playerId', \n    'battingOrder', \n    'gamesPlayedBatting', \n    'flyOuts',\n    'groundOuts', \n    'runsScored', \n    'doubles', \n    'triples', \n    'homeRuns',\n    'strikeOuts', \n    'baseOnBalls', \n    'intentionalWalks', \n    'hits', \n    'hitByPitch',\n    'atBats', \n    'caughtStealing', \n    'stolenBases', \n    'groundIntoDoublePlay',\n    'groundIntoTriplePlay', \n    'plateAppearances', \n    'totalBases', \n    'rbi',\n    'leftOnBase', \n    'sacBunts', \n    'sacFlies', \n    'catchersInterference',\n    'pickoffs', \n    'gamesPlayedPitching', \n    'gamesStartedPitching',\n    'completeGamesPitching', \n    'shutoutsPitching', \n    'winsPitching',\n    'lossesPitching', \n    'flyOutsPitching', \n    'airOutsPitching',\n    'groundOutsPitching', \n    'runsPitching', \n    'doublesPitching',\n    'triplesPitching', \n    'homeRunsPitching', \n    'strikeOutsPitching',\n    'baseOnBallsPitching', \n    'intentionalWalksPitching', \n    'hitsPitching',\n    'hitByPitchPitching', \n    'atBatsPitching', \n    'caughtStealingPitching',\n    'stolenBasesPitching', \n    'inningsPitched', \n    'saveOpportunities',\n    'earnedRuns', \n    'battersFaced', \n    'outsPitching', \n    'pitchesThrown', \n    'balls',\n    'strikes', \n    'hitBatsmen', \n    'balks', \n    'wildPitches', \n    'pickoffsPitching',\n    'rbiPitching', \n    'gamesFinishedPitching', \n    'inheritedRunners',\n    'inheritedRunnersScored', \n    'catchersInterferencePitching',\n    'sacBuntsPitching', \n    'sacFliesPitching', \n    'saves', \n    'holds', \n    'blownSaves',\n    'assists', \n    'putOuts', \n    'errors', \n    'chances', \n    'date'\n]\n\nawards_cols = [\n    'date', \n    'playerId',\n    'awardId'\n]\n\nplayerTwitterFollowers_cols = [\n    'playerId', \n    'numberOfFollowers'\n]\n\nteamTwitterFollowers_cols = [\n    'teamId', \n    'numberOfFollowers'\n]\n\nstandings_cols = [\n    'teamId', \n#     'wildCardRank', \n    'wins', \n    'losses', \n#     'divisionChamp', \n#     'divisionLeader', \n#     'wildCardLeader', \n    'lastTenWins',\n    'lastTenLosses',\n    'date'\n]\n\nfeature_cols = [\n    'label_playerId', \n    'label_primaryPositionName', \n    'label_teamId',\n    'label_status',\n    'playerId', \n    'battingOrder', \n    'gamesPlayedBatting', \n    'flyOuts',\n    'groundOuts', \n    'runsScored', \n    'doubles', \n    'triples', \n    'homeRuns',\n    'strikeOuts', \n    'baseOnBalls', \n    'intentionalWalks', \n    'hits', \n    'hitByPitch',\n    'atBats', \n    'caughtStealing', \n    'stolenBases', \n    'groundIntoDoublePlay',\n    'groundIntoTriplePlay', \n    'plateAppearances', \n    'totalBases', \n    'rbi',\n    'leftOnBase', \n    'sacBunts', \n    'sacFlies', \n    'catchersInterference',\n    'pickoffs', \n    'gamesPlayedPitching', \n    'gamesStartedPitching',\n    'completeGamesPitching', \n    'shutoutsPitching', \n    'winsPitching',\n    'lossesPitching', \n    'flyOutsPitching', \n    'airOutsPitching',\n    'groundOutsPitching', \n    'runsPitching', \n    'doublesPitching',\n    'triplesPitching', \n    'homeRunsPitching', \n    'strikeOutsPitching',\n    'baseOnBallsPitching', \n    'intentionalWalksPitching', \n    'hitsPitching',\n    'hitByPitchPitching', \n    'atBatsPitching', \n    'caughtStealingPitching',\n    'stolenBasesPitching', \n    'inningsPitched', \n    'saveOpportunities',\n    'earnedRuns', \n    'battersFaced', \n    'outsPitching', \n    'pitchesThrown', \n    'balls',\n    'strikes', \n    'hitBatsmen', \n    'balks', \n    'wildPitches', \n    'pickoffsPitching',\n    'rbiPitching', \n    'gamesFinishedPitching', \n    'inheritedRunners',\n    'inheritedRunnersScored', \n    'catchersInterferencePitching',\n    'sacBuntsPitching', \n    'sacFliesPitching', \n    'saves', \n    'holds', \n    'blownSaves',\n    'assists', \n    'putOuts', \n    'errors', \n    'chances', \n    'target1_mean',\n    'target1_median',\n    'target1_std',\n    'target1_min',\n    'target1_max',\n    'target1_prob',\n    'target2_mean',\n    'target2_median',\n    'target2_std',\n    'target2_min',\n    'target2_max',\n    'target2_prob',\n    'target3_mean',\n    'target3_median',\n    'target3_std',\n    'target3_min',\n    'target3_max',\n    'target3_prob',\n    'target4_mean',\n    'target4_median',\n    'target4_std',\n    'target4_min',\n    'target4_max',\n    'target4_prob',\n    'awardId_count',\n    'playernumberOfFollowers',               \n    'teamnumberOfFollowers',\n    'label_leagueId',\n    'label_divisionId',\n    'wins', \n    'losses', \n    'lastTenWins',\n    'lastTenLosses'\n]","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:35.83323Z","iopub.execute_input":"2021-07-31T04:08:35.833657Z","iopub.status.idle":"2021-07-31T04:08:35.856254Z","shell.execute_reply.started":"2021-07-31T04:08:35.833618Z","shell.execute_reply":"2021-07-31T04:08:35.854558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data and groupby","metadata":{}},{"cell_type":"code","source":"players = pd.read_csv(BASE_DIR / 'players.csv', usecols = players_cols)\nplayers = reduce_mem_usage(players)\n\n\nteams = pd.read_csv(BASE_DIR / 'teams.csv', usecols = teams_cols)\nteams = teams.rename(columns = {'id':'teamId'})\nteams = reduce_mem_usage(teams)\n\n\nrosters = pd.read_csv(TRAIN_DIR / 'rosters_train.csv', usecols = rosters_cols)\nrosters = reduce_mem_usage(rosters)\n\n\ntargets = pd.read_csv(TRAIN_DIR / 'nextDayPlayerEngagement_train.csv', usecols = targets_cols)\ntargets = reduce_mem_usage(targets)\n\n\nscores = pd.read_csv(TRAIN_DIR / 'playerBoxScores_train.csv', usecols = scores_cols)\nscores = scores.groupby(['playerId', 'date']).sum().reset_index()\nscores = reduce_mem_usage(scores)\n\n\nawards = pd.read_csv(TRAIN_DIR / 'awards_train.csv', usecols = awards_cols)\n# awards = awards.groupby(['playerId', 'date']).count().reset_index()\n\n\nawards_count = awards[['playerId', 'awardId']].groupby('playerId').count().reset_index()\nawards_count = awards_count.rename(columns = {'awardId':'awardId_count'})\nawards_count = reduce_mem_usage(awards_count)\n\n\nplayerTwitterFollowers = pd.read_csv(TRAIN_DIR / 'playerTwitterFollowers_train.csv', usecols = playerTwitterFollowers_cols)\nplayerTwitterFollowers = playerTwitterFollowers.groupby('playerId').sum().reset_index()\nplayerTwitterFollowers = playerTwitterFollowers.rename(columns = {'numberOfFollowers':'playernumberOfFollowers'})\nplayerTwitterFollowers = reduce_mem_usage(playerTwitterFollowers)\n\n\nteamTwitterFollowers = pd.read_csv(TRAIN_DIR / 'teamTwitterFollowers_train.csv', usecols = teamTwitterFollowers_cols)\nteamTwitterFollowers = teamTwitterFollowers.groupby('teamId').sum().reset_index()\nteamTwitterFollowers = teamTwitterFollowers.rename(columns = {'numberOfFollowers':'teamnumberOfFollowers'})\nteamTwitterFollowers = reduce_mem_usage(teamTwitterFollowers)\n\n\nstandings = pd.read_csv(TRAIN_DIR / 'standings_train.csv', usecols = standings_cols)\nstandings = reduce_mem_usage(standings)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:39.178469Z","iopub.execute_input":"2021-07-31T04:08:39.178874Z","iopub.status.idle":"2021-07-31T04:08:49.990515Z","shell.execute_reply.started":"2021-07-31T04:08:39.178839Z","shell.execute_reply":"2021-07-31T04:08:49.98947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_target_stats = pd.read_csv(\"../input/player-target-stats/player_target_stats.csv\")\ndata_names=player_target_stats.columns.values.tolist()\ndata_names","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:53.667739Z","iopub.execute_input":"2021-07-31T04:08:53.668176Z","iopub.status.idle":"2021-07-31T04:08:53.70829Z","shell.execute_reply.started":"2021-07-31T04:08:53.66814Z","shell.execute_reply":"2021-07-31T04:08:53.707219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make train data","metadata":{}},{"cell_type":"code","source":"# creat dataset\n\ntrain = targets.copy()[targets_cols]\n\nprint(targets[targets_cols].shape)\n\ntrain = train.merge(\n    players, \n    on=['playerId'], \n    how='left'\n)\ngc.collect()\n\nprint(train.shape, 'after_players')\nprint('--------------------------------------')\n\ntrain = train.merge(\n    rosters, \n    on=['playerId', 'date'], \n    how='left'\n)\ngc.collect()\n\nprint(train.shape, 'after_rosters')\nprint('--------------------------------------')\n\ntrain = train.merge(\n    scores, \n    on=['playerId', 'date'], \n    how='left'\n)\ngc.collect()\n\nprint(train.shape, 'after_scores')\nprint('--------------------------------------')\n\ntrain = train.merge(\n    player_target_stats, \n    how='inner', \n    on= \"playerId\",\n)\ngc.collect()\n\nprint(train.shape, 'after_player_target_stats')\n\n\nprint('--------------------------------------')\n\ntrain = train.merge(\n    teams,\n    on = 'teamId',\n    how='left'\n)\n# del rosters\ngc.collect()\n\nprint(train.shape, 'after_teams')\nprint('--------------------------------------')\n\ntrain = train.merge(\n    awards_count,\n    on = 'playerId',\n    how = 'left'\n)\n\ntrain['awardId_count'] = train['awardId_count'].fillna(0)\n\nprint(train.shape, 'after_awards_count')\nprint('--------------------------------------')\n\ntrain = train.merge(\n    playerTwitterFollowers, \n    how = 'left', \n    on = 'playerId'\n)\ngc.collect()\n\nprint(train.shape, 'after_playerTwitter')\nprint('--------------------------------------')\n\n\ntrain = train.merge(\n    teamTwitterFollowers, \n    how = 'left', \n    on = 'teamId'\n)\ngc.collect()\n\nprint(train.shape, 'after_taemTwitter')\nprint('--------------------------------------')\n\ntrain = train.merge(\n    standings, \n    how = 'left', \n    on = ['teamId', 'date']\n)\ngc.collect()\n\nprint(train.shape, 'after_standings')\nprint('--------------------------------------')\n\n\n# label encoding\nplayer2num = {c: i for i, c in enumerate(train['playerId'].unique())}\nposition2num = {c: i for i, c in enumerate(train['primaryPositionName'].unique())}\nteamid2num = {c: i for i, c in enumerate(train['teamId'].unique())}\nstatus2num = {c: i for i, c in enumerate(train['status'].unique())}\nleagueId2num = {c: i for i, c in enumerate(train['leagueId'].unique())}\ndivisionId2num = {c: i for i, c in enumerate(train['divisionId'].unique())}\n\n\ntrain['label_playerId'] = train['playerId'].map(player2num)\ntrain['label_primaryPositionName'] = train['primaryPositionName'].map(position2num)\ntrain['label_teamId'] = train['teamId'].map(teamid2num)\ntrain['label_status'] = train['status'].map(status2num)\ntrain['label_leagueId'] = train['leagueId'].map(leagueId2num)\ntrain['label_divisionId'] = train['divisionId'].map(divisionId2num)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:08:57.193766Z","iopub.execute_input":"2021-07-31T04:08:57.194393Z","iopub.status.idle":"2021-07-31T04:09:18.583198Z","shell.execute_reply.started":"2021-07-31T04:08:57.194349Z","shell.execute_reply":"2021-07-31T04:09:18.582297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:38:29.001673Z","iopub.execute_input":"2021-07-31T03:38:29.0021Z","iopub.status.idle":"2021-07-31T03:38:29.024501Z","shell.execute_reply.started":"2021-07-31T03:38:29.002061Z","shell.execute_reply":"2021-07-31T03:38:29.023434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:38:29.025756Z","iopub.execute_input":"2021-07-31T03:38:29.026061Z","iopub.status.idle":"2021-07-31T03:38:30.712489Z","shell.execute_reply.started":"2021-07-31T03:38:29.026027Z","shell.execute_reply":"2021-07-31T03:38:30.711558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Divide train and valid data","metadata":{}},{"cell_type":"code","source":"train_X = train[feature_cols]\ntrain_y = train[['target1', 'target2', 'target3', 'target4']]\n\n_index = (train['date'] < 20210401)\nx_train = train_X.loc[_index].reset_index(drop=True)\ny_train = train_y.loc[_index].reset_index(drop=True)\nx_valid = train_X.loc[~_index].reset_index(drop=True)\ny_valid = train_y.loc[~_index].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:17:17.434324Z","iopub.execute_input":"2021-07-31T04:17:17.434963Z","iopub.status.idle":"2021-07-31T04:17:21.36688Z","shell.execute_reply.started":"2021-07-31T04:17:17.434926Z","shell.execute_reply":"2021-07-31T04:17:21.365947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:19:24.545811Z","iopub.execute_input":"2021-07-31T04:19:24.54627Z","iopub.status.idle":"2021-07-31T04:19:24.556937Z","shell.execute_reply.started":"2021-07-31T04:19:24.546235Z","shell.execute_reply":"2021-07-31T04:19:24.555893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle as pl\nf=open('x_train.pkl','wb')\nf1=open('y_train.pkl','wb') \nf2=open('x_valid.pkl','wb') \nf3=open('y_valid.pkl','wb') \npl.dump(x_train,f)\npl.dump(y_train,f1)\npl.dump(x_valid,f2)\npl.dump(y_valid,f3)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T03:46:30.081432Z","iopub.execute_input":"2021-07-31T03:46:30.081865Z","iopub.status.idle":"2021-07-31T03:46:32.328179Z","shell.execute_reply.started":"2021-07-31T03:46:30.08183Z","shell.execute_reply":"2021-07-31T03:46:32.327134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Example for tuning","metadata":{}},{"cell_type":"markdown","source":"## Predict","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}