{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pickle\nimport gc\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom catboost import CatBoostRegressor\nfrom catboost import Pool\n\npd.set_option('display.max_columns', 100)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-23T12:56:53.328312Z","iopub.execute_input":"2021-07-23T12:56:53.330463Z","iopub.status.idle":"2021-07-23T12:56:56.032712Z","shell.execute_reply.started":"2021-07-23T12:56:53.330339Z","shell.execute_reply":"2021-07-23T12:56:56.031666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mydata_dir = '../input/mlbiyanagatargetencseasons/'\ntrain_dir = '../input/mlb-pdef-train-dataset/'","metadata":{"execution":{"iopub.status.busy":"2021-07-23T12:57:02.311803Z","iopub.execute_input":"2021-07-23T12:57:02.312170Z","iopub.status.idle":"2021-07-23T12:57:02.316313Z","shell.execute_reply.started":"2021-07-23T12:57:02.312140Z","shell.execute_reply":"2021-07-23T12:57:02.315247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv('../input/mlb-player-digital-engagement-forecasting/players.csv')\nrosters = pd.read_pickle(train_dir + 'rosters_train.pkl')\nscores = pd.read_pickle(train_dir + 'playerBoxScores_train.pkl')\n\nwith open(mydata_dir + 'player2num.pkl', 'rb') as f:\n    player2num = pickle.load(f)\nwith open(mydata_dir + 'position2num.pkl', 'rb') as f:\n    position2num = pickle.load(f)\nwith open(mydata_dir + 'teamid2num.pkl', 'rb') as f:\n    teamid2num = pickle.load(f)\nwith open(mydata_dir + 'status2num.pkl', 'rb') as f:\n    status2num = pickle.load(f)\n    \nwith open(mydata_dir + 'model1_cat.pkl', 'rb') as f:\n    model1 = pickle.load(f)\nwith open(mydata_dir + 'model2_cat.pkl', 'rb') as f:\n    model2 = pickle.load(f)\nwith open(mydata_dir + 'model3_cat.pkl', 'rb') as f:\n    model3 = pickle.load(f)\nwith open(mydata_dir + 'model4_cat.pkl', 'rb') as f:\n    model4 = pickle.load(f)\n\ntarget_stat_df = pd.read_pickle(mydata_dir + 'target_stat_df.pkl')","metadata":{"execution":{"iopub.status.busy":"2021-07-23T12:57:02.646752Z","iopub.execute_input":"2021-07-23T12:57:02.647111Z","iopub.status.idle":"2021-07-23T12:57:06.412316Z","shell.execute_reply.started":"2021-07-23T12:57:02.647074Z","shell.execute_reply":"2021-07-23T12:57:06.410931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mlb\nenv = mlb.make_env()\niter_test = env.iter_test()\n\ntargets_cols = ['playerId', 'target1', 'target2', 'target3', 'target4', 'date']\nplayers_cols = ['playerId', 'primaryPositionName']\nrosters_cols = ['playerId', 'teamId', 'status']\nscores_cols = ['playerId', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n               'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n               'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n               'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n               'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n               'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n               'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n               'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n               'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n               'groundOutsPitching', 'runsPitching', 'doublesPitching',\n               'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n               'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n               'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n               'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n               'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n               'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n               'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n               'inheritedRunnersScored', 'catchersInterferencePitching',\n               'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n               'assists', 'putOuts', 'errors', 'chances']\ntarget_stat_cols = ['playerId', 'target1_mean', 'target1_median', 'target1_std', 'target1_max', 'target1_min', \n                    'target2_mean', 'target2_median', 'target2_std', 'target2_max', 'target2_min', \n                    'target3_mean', 'target3_median', 'target3_std', 'target3_max', 'target3_min', \n                    'target4_mean', 'target4_median', 'target4_std', 'target4_max', 'target4_min']\nfeature_cols = ['label_playerId', 'label_primaryPositionName', 'label_teamId',\n                'label_status', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n                'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n                'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n                'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n                'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n                'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n                'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n                'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n                'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n                'groundOutsPitching', 'runsPitching', 'doublesPitching',\n                'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n                'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n                'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n                'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n                'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n                'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n                'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n                'inheritedRunnersScored', 'catchersInterferencePitching',\n                'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n                'assists', 'putOuts', 'errors', 'chances',\n                'target1_mean', 'target1_median', 'target1_std', 'target1_max', 'target1_min', \n                'target2_mean', 'target2_median', 'target2_std', 'target2_max', 'target2_min', \n                'target3_mean', 'target3_median', 'target3_std', 'target3_max', 'target3_min', \n                'target4_mean', 'target4_median', 'target4_std', 'target4_max', 'target4_min']\n\n# これ入れないとdf読み込みでエラー吐きます\nnull = np.nan\ntrue = True\nfalse = False\n\n    \nfor (test_df, sample_prediction_df) in iter_test:\n    \n    sample_prediction_df = sample_prediction_df.reset_index(drop=True)\n    \n    # creat dataset\n    sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    # Dealing with missing values\n    if test_df['rosters'].iloc[0] == test_df['rosters'].iloc[0]:\n        test_rosters = pd.DataFrame(eval(test_df['rosters'].iloc[0]))\n    else:\n        test_rosters = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in rosters.columns:\n            if col == 'playerId': continue\n            test_rosters[col] = np.nan\n            \n    if test_df['playerBoxScores'].iloc[0] == test_df['playerBoxScores'].iloc[0]:\n        test_scores = pd.DataFrame(eval(test_df['playerBoxScores'].iloc[0]))\n    else:\n        test_scores = pd.DataFrame({'playerId': sample_prediction_df['playerId']})\n        for col in scores.columns:\n            if col == 'playerId': continue\n            test_scores[col] = np.nan\n            \n    test_scores = test_scores.groupby('playerId').sum().reset_index()\n    \n    test = sample_prediction_df[['playerId']].copy()\n    test = test.merge(players[players_cols], on='playerId', how='left')\n    test = test.merge(test_rosters[rosters_cols], on='playerId', how='left')\n    test = test.merge(test_scores[scores_cols], on='playerId', how='left')\n    test = test.merge(target_stat_df, how='inner', left_on=[\"playerId\"],right_on=[\"playerId\"])\n\n    test['label_playerId'] = test['playerId'].map(player2num)\n    test['label_primaryPositionName'] = test['primaryPositionName'].map(position2num)\n    test['label_teamId'] = test['teamId'].map(teamid2num)\n    test['label_status'] = test['status'].map(status2num)\n    \n    test_X = test[feature_cols]\n    \n    # predict\n    pred1 = model1.predict(test_X)\n    pred2 = model2.predict(test_X)\n    pred3 = model3.predict(test_X)\n    pred4 = model4.predict(test_X)\n    \n    # merge submission\n    sample_prediction_df['target1'] = np.clip(pred1, 0, 100)\n    sample_prediction_df['target2'] = np.clip(pred2, 0, 100)\n    sample_prediction_df['target3'] = np.clip(pred3, 0, 100)\n    sample_prediction_df['target4'] = np.clip(pred4, 0, 100)\n    sample_prediction_df = sample_prediction_df.fillna(0.)\n    del sample_prediction_df['playerId']\n    \n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T12:57:24.541184Z","iopub.execute_input":"2021-07-23T12:57:24.541621Z","iopub.status.idle":"2021-07-23T12:57:26.733029Z","shell.execute_reply.started":"2021-07-23T12:57:24.541575Z","shell.execute_reply":"2021-07-23T12:57:26.731819Z"},"trusted":true},"execution_count":null,"outputs":[]}]}