{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport time\nimport mlb\nimport gc\npd.set_option(\"display.max_rows\", None, \"display.max_columns\", None)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-22T18:41:34.919532Z","iopub.execute_input":"2021-06-22T18:41:34.919959Z","iopub.status.idle":"2021-06-22T18:41:34.965324Z","shell.execute_reply.started":"2021-06-22T18:41:34.919865Z","shell.execute_reply":"2021-06-22T18:41:34.964394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_path = r\"../input/mlb-player-digital-engagement-forecasting\"\npreload_path = r\"../input/k/ssmohanty/mlb-master-ads-creation-v1-basic-eda\"\n\n#---------------------------------------------------------------------------\nmaster_ads = pd.read_csv(os.path.join(preload_path, \"master_ads_4_left_join.csv\"))\nmaster_ads.drop(columns=['Unnamed: 0','numberOfFollowers_x','numberOfFollowers_y'],inplace=True)\nprint(master_ads.shape)\nprint(master_ads.info())\nmaster_ads.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:41:35.047280Z","iopub.execute_input":"2021-06-22T18:41:35.047649Z","iopub.status.idle":"2021-06-22T18:41:57.656426Z","shell.execute_reply.started":"2021-06-22T18:41:35.047618Z","shell.execute_reply":"2021-06-22T18:41:57.655581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# master_ads['trace_back_flag'] = np.where(master_ads.notna().all(axis=1), 1, 0)\nmaster_ads = master_ads.fillna(0)\nmaster_ads.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:41:57.657582Z","iopub.execute_input":"2021-06-22T18:41:57.657975Z","iopub.status.idle":"2021-06-22T18:41:59.608917Z","shell.execute_reply.started":"2021-06-22T18:41:57.657947Z","shell.execute_reply":"2021-06-22T18:41:59.608052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"master_ads.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:41:59.610354Z","iopub.execute_input":"2021-06-22T18:41:59.610819Z","iopub.status.idle":"2021-06-22T18:42:00.678563Z","shell.execute_reply.started":"2021-06-22T18:41:59.610762Z","shell.execute_reply":"2021-06-22T18:42:00.677707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# master_ads[master_ads['trace_back_flag']==1].head()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:00.679870Z","iopub.execute_input":"2021-06-22T18:42:00.680326Z","iopub.status.idle":"2021-06-22T18:42:00.683749Z","shell.execute_reply.started":"2021-06-22T18:42:00.680295Z","shell.execute_reply":"2021-06-22T18:42:00.682796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dateCols = [col for col in master_ads.columns if ('date' in col) | ('Date' in col)]\ndateCols","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:00.684948Z","iopub.execute_input":"2021-06-22T18:42:00.685261Z","iopub.status.idle":"2021-06-22T18:42:00.699321Z","shell.execute_reply.started":"2021-06-22T18:42:00.685231Z","shell.execute_reply":"2021-06-22T18:42:00.698338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"master_ads[['playerId','eng_date_pre','eng_date']].tail()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:00.700811Z","iopub.execute_input":"2021-06-22T18:42:00.701165Z","iopub.status.idle":"2021-06-22T18:42:00.816075Z","shell.execute_reply.started":"2021-06-22T18:42:00.701108Z","shell.execute_reply":"2021-06-22T18:42:00.815356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"master_ads[['target1','target2','target3','target4']].describe()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:00.817136Z","iopub.execute_input":"2021-06-22T18:42:00.817506Z","iopub.status.idle":"2021-06-22T18:42:01.303995Z","shell.execute_reply.started":"2021-06-22T18:42:00.817479Z","shell.execute_reply":"2021-06-22T18:42:01.303202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_1_med = master_ads['target1'].mean()\ntarget_2_med = master_ads['target2'].mean()\ntarget_3_med = master_ads['target3'].mean()\ntarget_4_med = master_ads['target4'].mean()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:01.306044Z","iopub.execute_input":"2021-06-22T18:42:01.306520Z","iopub.status.idle":"2021-06-22T18:42:01.341602Z","shell.execute_reply.started":"2021-06-22T18:42:01.306488Z","shell.execute_reply":"2021-06-22T18:42:01.340762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Necessary libraries for ML pred","metadata":{}},{"cell_type":"code","source":"from sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:01.343215Z","iopub.execute_input":"2021-06-22T18:42:01.343630Z","iopub.status.idle":"2021-06-22T18:42:02.405900Z","shell.execute_reply.started":"2021-06-22T18:42:01.343600Z","shell.execute_reply":"2021-06-22T18:42:02.405167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train-Test Split","metadata":{}},{"cell_type":"code","source":"target_cols = ['target1','target2','target3','target4']\ntotal_cols = list(master_ads.columns)\nX_cols = [col for col in total_cols if col not in target_cols]\n\nprint('Total Cols :',len(total_cols))\nprint('X Cols :',len(X_cols))\n#--------------------------------------------------------------------------------------------\n\nX = master_ads.loc[:, X_cols]\ny = master_ads.loc[:, target_cols]\n\n#------------------------------------------------------------------------------------------\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.03,test_size=0.15, random_state=100)\n\nprint('Train X shape :',X_train.shape)\nprint('Test X shape :',X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:02.406962Z","iopub.execute_input":"2021-06-22T18:42:02.407406Z","iopub.status.idle":"2021-06-22T18:42:04.524056Z","shell.execute_reply.started":"2021-06-22T18:42:02.407374Z","shell.execute_reply":"2021-06-22T18:42:04.522723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check for un-encoded categorical cols, date columns, Id columns","metadata":{}},{"cell_type":"code","source":"catCols = [col for col in X_train.columns if X_train[col].dtype==\"O\"]\nlen(catCols)\ncatCols\n\ndateCols = [col for col in X_train.columns if ('date' in col) | ('Date' in col)]\nlen(dateCols)\n\nIdCols = [col for col in X_train.columns if ('Id' in col) | ('id' in col)]\nlen(IdCols)\nIdCols\n\nprint(dateCols)\nprint(IdCols)\nprint(catCols)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:04.525521Z","iopub.execute_input":"2021-06-22T18:42:04.525870Z","iopub.status.idle":"2021-06-22T18:42:04.538717Z","shell.execute_reply.started":"2021-06-22T18:42:04.525836Z","shell.execute_reply":"2021-06-22T18:42:04.537544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Further data processing (Removing Level Columns from train and test X)","metadata":{}},{"cell_type":"code","source":"level_cols = ['eng_date','eng_date_pre','playerId']\n\nX_train_1 = X_train.drop(columns=level_cols)\nX_test_1 = X_test.drop(columns=level_cols)\n\n#--------------------------------------------\nprint('Train X reduced shape :',X_train_1.shape)\nprint('Test X reduced shape :',X_test_1.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:04.540506Z","iopub.execute_input":"2021-06-22T18:42:04.540921Z","iopub.status.idle":"2021-06-22T18:42:04.729155Z","shell.execute_reply.started":"2021-06-22T18:42:04.540878Z","shell.execute_reply":"2021-06-22T18:42:04.728007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train_1 = X_train_1.sample(100000,random_state=100)\n\n\n#print('Train shrinked shape :',X_train_1.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:04.730783Z","iopub.execute_input":"2021-06-22T18:42:04.731269Z","iopub.status.idle":"2021-06-22T18:42:04.743568Z","shell.execute_reply.started":"2021-06-22T18:42:04.731223Z","shell.execute_reply":"2021-06-22T18:42:04.742250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_1.info()\nX_test_1.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:04.745417Z","iopub.execute_input":"2021-06-22T18:42:04.745954Z","iopub.status.idle":"2021-06-22T18:42:04.784229Z","shell.execute_reply.started":"2021-06-22T18:42:04.745920Z","shell.execute_reply":"2021-06-22T18:42:04.782995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train\ndel X_test\n#del master_ads\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:04.785685Z","iopub.execute_input":"2021-06-22T18:42:04.786093Z","iopub.status.idle":"2021-06-22T18:42:04.900664Z","shell.execute_reply.started":"2021-06-22T18:42:04.786049Z","shell.execute_reply":"2021-06-22T18:42:04.899256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ['awardPlayerTeamId', 'awardSeason', 'fromTeamId', 'toTeamId']\nX_train_1.drop(columns=['awardPlayerTeamId', 'awardSeason', 'fromTeamId', 'toTeamId'],inplace=True)\nX_test_1.drop(columns=['awardPlayerTeamId', 'awardSeason', 'fromTeamId', 'toTeamId'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:04.902333Z","iopub.execute_input":"2021-06-22T18:42:04.902909Z","iopub.status.idle":"2021-06-22T18:42:05.032974Z","shell.execute_reply.started":"2021-06-22T18:42:04.902711Z","shell.execute_reply":"2021-06-22T18:42:05.031981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_1.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:05.034280Z","iopub.execute_input":"2021-06-22T18:42:05.034587Z","iopub.status.idle":"2021-06-22T18:42:05.060691Z","shell.execute_reply.started":"2021-06-22T18:42:05.034558Z","shell.execute_reply":"2021-06-22T18:42:05.059597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGB Regressor w/ multiple o/p's (sklearn wrapper)","metadata":{}},{"cell_type":"code","source":"xgb.XGBRegressor(random_state=100).get_params()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:05.061881Z","iopub.execute_input":"2021-06-22T18:42:05.062176Z","iopub.status.idle":"2021-06-22T18:42:05.070348Z","shell.execute_reply.started":"2021-06-22T18:42:05.062147Z","shell.execute_reply":"2021-06-22T18:42:05.069163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from xgboost import xgb\n\n#--------------------------------------------------\nestimator = xgb.XGBRegressor(reg_alpha = 50, reg_almbda = 60,random_state=100,n_jobs=-1)\nmodel = MultiOutputRegressor(estimator, n_jobs=-1)\n\n#--------------------------------------------------\nmodel.fit(X_train_1, y_train)\n\n#--------------------------------------------------\npred_xgb_multi = model.predict(X_test_1)\n               \n\n#--------------------------------------------------\npred_xgb_multi.shape ","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:42:05.072049Z","iopub.execute_input":"2021-06-22T18:42:05.072453Z","iopub.status.idle":"2021-06-22T18:43:06.103010Z","shell.execute_reply.started":"2021-06-22T18:42:05.072423Z","shell.execute_reply":"2021-06-22T18:43:06.101948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scoring(truth_array,pred_array):\n    \n    diff = abs(truth_array - pred_array)\n    \n    print(diff[0:3])\n    \n    mean_col_wise = np.mean(diff,axis=0)\n    print(mean_col_wise)\n    \n    mean_MAE = np.mean(mean_col_wise)\n    \n    return mean_MAE","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.104398Z","iopub.execute_input":"2021-06-22T18:43:06.104687Z","iopub.status.idle":"2021-06-22T18:43:06.109895Z","shell.execute_reply.started":"2021-06-22T18:43:06.104658Z","shell.execute_reply":"2021-06-22T18:43:06.108897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(y_test)[0:3]","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.111556Z","iopub.execute_input":"2021-06-22T18:43:06.112263Z","iopub.status.idle":"2021-06-22T18:43:06.136673Z","shell.execute_reply.started":"2021-06-22T18:43:06.112217Z","shell.execute_reply":"2021-06-22T18:43:06.135266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_xgb_multi[0:3]","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.138108Z","iopub.execute_input":"2021-06-22T18:43:06.138446Z","iopub.status.idle":"2021-06-22T18:43:06.144682Z","shell.execute_reply.started":"2021-06-22T18:43:06.138416Z","shell.execute_reply":"2021-06-22T18:43:06.143625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_score = scoring(np.array(y_test),pred_xgb_multi)\nxgb_score","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.149378Z","iopub.execute_input":"2021-06-22T18:43:06.149706Z","iopub.status.idle":"2021-06-22T18:43:06.180763Z","shell.execute_reply.started":"2021-06-22T18:43:06.149677Z","shell.execute_reply":"2021-06-22T18:43:06.179716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_test = pd.read_csv(os.path.join(main_path, \"example_test.csv\"))\nexample_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.182596Z","iopub.execute_input":"2021-06-22T18:43:06.182877Z","iopub.status.idle":"2021-06-22T18:43:06.895449Z","shell.execute_reply.started":"2021-06-22T18:43:06.182850Z","shell.execute_reply":"2021-06-22T18:43:06.894297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unpack_json(json_str):\n    return np.nan if pd.isna(json_str) else pd.read_json(json_str)\n\ndef data_extractor(df_,col_):\n    \n    print('------- Data Extraction for :',col_,'-------')\n    \n    final_df = pd.DataFrame()\n    \n    tmp = df_[col_]\n    tmp = tmp.dropna()\n    \n    #-------------------------------------------------------\n    for i in range(len(tmp)):\n        tmpdf = unpack_json(tmp.iloc[i])\n        final_df = final_df.append(tmpdf)\n    \n    #-------------------------------------------------------\n    print('Shape of final df BEFORE:',final_df.shape)\n    final_df = final_df.drop_duplicates()\n    print('Shape of final df AFTER:',final_df.shape)\n    print(final_df.columns)\n    #-------------------------------------------------------\n    \n    return final_df\n\n\ndef label_encoding_test(unique,col,df):\n    \n    print('-----',col,'-----')\n    \n    print('Before :',len(unique))\n    encodes_pre = unique\n    \n    encodes = np.arange(len(encodes_pre))\n    encodes = list(encodes+1)\n\n    mapping = dict(zip(encodes_pre,encodes))\n    df = df.replace({col:mapping})\n    print('After :',df[col].nunique())\n    \n    return df\n\ndef date_type_converter(df):\n    \n    date_cols = [col for col in list(df.columns) if (('date' in col) | ('Date' in col))]\n    \n    for col in date_cols:\n        df[col] = pd.to_datetime(df[col], format='%Y-%m-%d')\n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.896861Z","iopub.execute_input":"2021-06-22T18:43:06.897176Z","iopub.status.idle":"2021-06-22T18:43:06.908527Z","shell.execute_reply.started":"2021-06-22T18:43:06.897145Z","shell.execute_reply":"2021-06-22T18:43:06.907539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_test.columns","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.909685Z","iopub.execute_input":"2021-06-22T18:43:06.910016Z","iopub.status.idle":"2021-06-22T18:43:06.933138Z","shell.execute_reply.started":"2021-06-22T18:43:06.909987Z","shell.execute_reply":"2021-06-22T18:43:06.931865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.934925Z","iopub.execute_input":"2021-06-22T18:43:06.935449Z","iopub.status.idle":"2021-06-22T18:43:06.949891Z","shell.execute_reply.started":"2021-06-22T18:43:06.935404Z","shell.execute_reply":"2021-06-22T18:43:06.948902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = data_extractor(example_test,'teamTwitterFollowers')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.951249Z","iopub.execute_input":"2021-06-22T18:43:06.951537Z","iopub.status.idle":"2021-06-22T18:43:06.966212Z","shell.execute_reply.started":"2021-06-22T18:43:06.951508Z","shell.execute_reply":"2021-06-22T18:43:06.965370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transactions_features_ = ['playerId', 'date', 'fromTeamId', 'toTeamId','typeCode','typeDesc']\n# team_twitter_features = ['date', 'teamId','twitterHandle',\n#        'numberOfFollowers']#twitterHandle_x will be of team\n\n#['awardPlayerTeamId', 'awardSeason', 'fromTeamId', 'toTeamId']\n\nstandings_features = ['season', 'gameDate', 'divisionId', 'teamId', 'streakCode',\n       'divisionRank', 'leagueRank', 'wildCardRank', 'leagueGamesBack',\n       'sportGamesBack', 'divisionGamesBack', 'wins', 'losses', 'pct',\n       'runsAllowed', 'divisionChamp', 'divisionLeader',\n       'wildCardLeader', 'eliminationNumber', 'wildCardEliminationNumber',\n       'homeWins', 'homeLosses', 'awayWins', 'awayLosses', 'lastTenWins',\n       'lastTenLosses', 'extraInningWins', 'extraInningLosses', 'oneRunWins',\n       'oneRunLosses', 'dayWins', 'dayLosses', 'nightWins', 'nightLosses',\n       'grassWins', 'grassLosses', 'turfWins', 'turfLosses', 'divWins',\n       'divLosses', 'alWins', 'alLosses', 'nlWins', 'nlLosses', 'xWinLossPct']\n# player_twitter_features = ['date', 'playerId', 'twitterHandle',\n#        'numberOfFollowers'] #twitterHandle_y will be of player\nplayer_scores_features = ['home', 'gamePk', 'gameDate', 'teamId',\n       'playerId', 'jerseyNum', 'positionCode',\n       'positionType', 'battingOrder', 'gamesPlayedBatting', 'flyOuts',\n       'groundOuts', 'runsScored', 'doubles', 'triples', 'homeRuns',\n       'strikeOuts', 'baseOnBalls', 'intentionalWalks', 'hits', 'hitByPitch',\n       'atBats', 'caughtStealing', 'stolenBases', 'groundIntoDoublePlay',\n       'groundIntoTriplePlay', 'plateAppearances', 'totalBases', 'rbi',\n       'leftOnBase', 'sacBunts', 'sacFlies', 'catchersInterference',\n       'pickoffs', 'gamesPlayedPitching', 'gamesStartedPitching',\n       'completeGamesPitching', 'shutoutsPitching', 'winsPitching',\n       'lossesPitching', 'flyOutsPitching', 'airOutsPitching',\n       'groundOutsPitching', 'runsPitching', 'doublesPitching',\n       'triplesPitching', 'homeRunsPitching', 'strikeOutsPitching',\n       'baseOnBallsPitching', 'intentionalWalksPitching', 'hitsPitching',\n       'hitByPitchPitching', 'atBatsPitching', 'caughtStealingPitching',\n       'stolenBasesPitching', 'inningsPitched', 'saveOpportunities',\n       'earnedRuns', 'battersFaced', 'outsPitching', 'pitchesThrown', 'balls',\n       'strikes', 'hitBatsmen', 'balks', 'wildPitches', 'pickoffsPitching',\n       'rbiPitching', 'gamesFinishedPitching', 'inheritedRunners',\n       'inheritedRunnersScored', 'catchersInterferencePitching',\n       'sacBuntsPitching', 'sacFliesPitching', 'saves', 'holds', 'blownSaves',\n       'assists', 'putOuts', 'errors', 'chances']\n# awards_df_features = ['awardId', 'awardDate', 'awardSeason', 'playerId', 'awardPlayerTeamId']","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.967748Z","iopub.execute_input":"2021-06-22T18:43:06.968382Z","iopub.status.idle":"2021-06-22T18:43:06.980551Z","shell.execute_reply.started":"2021-06-22T18:43:06.968338Z","shell.execute_reply":"2021-06-22T18:43:06.979331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfinal_features = list(X_train_1.columns)\n\n#--------------------------------------------------------------------------------------------------------------------\ndef ads_generator(df,final_features_generic):\n    \n    to_extract = ['standings','playerBoxScores']\n    \n    start = time.time()\n    \n    df_final = data_extractor(df,'playerBoxScores')\n    df_final = df_final[player_scores_features]\n    df_final = date_type_converter(df_final)\n    \n    to_extract.remove('playerBoxScores')\n    \n    for col in to_extract:\n        \n        df_temp = data_extractor(df,col)\n        \n#         if col == 'transactions':\n#             qc_df = df_temp.copy()\n        \n        df_temp = date_type_converter(df_temp)\n        \n        if 'date' in list(df_temp.columns):\n            print('1')\n        \n        \n        if col == 'transactions':\n            print('---------------------- Merging Operation for :',col,'------------------------')\n            df_final = pd.merge(df_final,df_temp[transactions_features_],\n                                left_on=['playerId','gameDate'],\n                                right_on=['playerId','date'],how='left')\n        \n        elif col == 'teamTwitterFollowers':\n            print('---------------------- Merging Operation for :',col,'------------------------')\n            df_final = pd.merge(df_final,df_temp[team_twitter_features],\n                                    left_on=['teamId','gameDate'],\n                                    right_on=['teamId','date'],how='left')\n        elif col == 'standings':\n            print('---------------------- Merging Operation for :',col,'------------------------')\n            df_final = pd.merge(df_final,df_temp[standings_features],\n                                    left_on=['teamId','gameDate'],\n                                    right_on=['teamId','gameDate'],how='left')\n            \n        elif col == 'playerTwitterFollowers':\n            print('---------------------- Merging Operation for :',col,'------------------------')\n            df_final = pd.merge(df_final,df_temp[player_twitter_features],left_on=['playerId','gameDate'],\n                                right_on=['playerId','date'],how='left')\n        \n        elif col == 'awards_df':\n            print('---------------------- Merging Operation for :',col,'------------------------')\n            df_final = pd.merge(df_final,df_temp[awards_df_features],left_on=['playerId','gameDate'],\n                                right_on=['playerId','awardDate'],how='left')\n            \n    df_final = df_final.fillna(0)\n    \n    #----------------------------------------------------------------------------------------------------\n#     df_final = label_encoding_test(twitterHandle_x_unique,'twitterHandle_x',df_final)\n#     df_final = label_encoding_test(awards_df_unique,'awards_df',df_final)\n#     df_final = label_encoding_test(twitterHandle_y_unique,'twitterHandle_y',df_final)\n    \n    #----------------------------------------------------------------------------------------------------\n    level_cols_test = ['gameDate']\n    df_final_1 = df_final.drop(columns=level_cols_test)\n    \n    df_final_1 = df_final_1[final_features_generic]\n    \n    end = time.time()\n    print('Time to Run:',end-start)\n    \n    return df_final\n","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:06.982096Z","iopub.execute_input":"2021-06-22T18:43:06.982451Z","iopub.status.idle":"2021-06-22T18:43:07.003167Z","shell.execute_reply.started":"2021-06-22T18:43:06.982420Z","shell.execute_reply":"2021-06-22T18:43:07.002132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(example_test.iloc[0]).T","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.004431Z","iopub.execute_input":"2021-06-22T18:43:07.004939Z","iopub.status.idle":"2021-06-22T18:43:07.064078Z","shell.execute_reply.started":"2021-06-22T18:43:07.004901Z","shell.execute_reply":"2021-06-22T18:43:07.062899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(example_test.iloc[0]).T['date']","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.065195Z","iopub.execute_input":"2021-06-22T18:43:07.065635Z","iopub.status.idle":"2021-06-22T18:43:07.074021Z","shell.execute_reply.started":"2021-06-22T18:43:07.065604Z","shell.execute_reply":"2021-06-22T18:43:07.073310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#example_test\ntest_ads_gen_df = ads_generator(pd.DataFrame(example_test.iloc[0]).T,final_features)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.075015Z","iopub.execute_input":"2021-06-22T18:43:07.075432Z","iopub.status.idle":"2021-06-22T18:43:07.245674Z","shell.execute_reply.started":"2021-06-22T18:43:07.075403Z","shell.execute_reply":"2021-06-22T18:43:07.244574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ads_gen_df = test_ads_gen_df[list(X_train_1.columns)]\ntype(test_ads_gen_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.247131Z","iopub.execute_input":"2021-06-22T18:43:07.247529Z","iopub.status.idle":"2021-06-22T18:43:07.256266Z","shell.execute_reply.started":"2021-06-22T18:43:07.247487Z","shell.execute_reply":"2021-06-22T18:43:07.255175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ads_gen_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.257922Z","iopub.execute_input":"2021-06-22T18:43:07.258374Z","iopub.status.idle":"2021-06-22T18:43:07.375446Z","shell.execute_reply.started":"2021-06-22T18:43:07.258330Z","shell.execute_reply":"2021-06-22T18:43:07.374261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_1.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.377090Z","iopub.execute_input":"2021-06-22T18:43:07.377527Z","iopub.status.idle":"2021-06-22T18:43:07.482988Z","shell.execute_reply.started":"2021-06-22T18:43:07.377484Z","shell.execute_reply":"2021-06-22T18:43:07.482109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds = model.predict(test_ads_gen_df)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.484313Z","iopub.execute_input":"2021-06-22T18:43:07.484740Z","iopub.status.idle":"2021-06-22T18:43:07.609661Z","shell.execute_reply.started":"2021-06-22T18:43:07.484697Z","shell.execute_reply":"2021-06-22T18:43:07.608858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds[:,0].shape","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.610710Z","iopub.execute_input":"2021-06-22T18:43:07.611015Z","iopub.status.idle":"2021-06-22T18:43:07.617021Z","shell.execute_reply.started":"2021-06-22T18:43:07.610986Z","shell.execute_reply":"2021-06-22T18:43:07.616286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"example_submission = pd.read_csv(os.path.join(main_path, \"example_sample_submission.csv\"))\nexample_submission.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.618253Z","iopub.execute_input":"2021-06-22T18:43:07.618710Z","iopub.status.idle":"2021-06-22T18:43:07.652014Z","shell.execute_reply.started":"2021-06-22T18:43:07.618677Z","shell.execute_reply":"2021-06-22T18:43:07.650991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(example_submission[0:2])","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.653380Z","iopub.execute_input":"2021-06-22T18:43:07.653684Z","iopub.status.idle":"2021-06-22T18:43:07.661688Z","shell.execute_reply.started":"2021-06-22T18:43:07.653653Z","shell.execute_reply":"2021-06-22T18:43:07.660452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = pd.read_csv(os.path.join(main_path, \"players.csv\"))\nplayers.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.662926Z","iopub.execute_input":"2021-06-22T18:43:07.663218Z","iopub.status.idle":"2021-06-22T18:43:07.704071Z","shell.execute_reply.started":"2021-06-22T18:43:07.663192Z","shell.execute_reply":"2021-06-22T18:43:07.703231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# type(players[players['playerId']>50000000000])","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.705551Z","iopub.execute_input":"2021-06-22T18:43:07.705952Z","iopub.status.idle":"2021-06-22T18:43:07.710199Z","shell.execute_reply.started":"2021-06-22T18:43:07.705910Z","shell.execute_reply":"2021-06-22T18:43:07.709238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x/hsybdksh","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.711485Z","iopub.execute_input":"2021-06-22T18:43:07.711780Z","iopub.status.idle":"2021-06-22T18:43:07.724486Z","shell.execute_reply.started":"2021-06-22T18:43:07.711752Z","shell.execute_reply":"2021-06-22T18:43:07.723391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try:\n#     x/hsybdksh\n# except:\n#     ax = 897\n    \n# print(ax)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.725978Z","iopub.execute_input":"2021-06-22T18:43:07.727000Z","iopub.status.idle":"2021-06-22T18:43:07.737336Z","shell.execute_reply.started":"2021-06-22T18:43:07.726949Z","shell.execute_reply":"2021-06-22T18:43:07.736276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submitting Predictions","metadata":{}},{"cell_type":"code","source":"env = mlb.make_env() # initialize the environment\niter_test = env.iter_test() # iterator which loops over each date in test set","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.738461Z","iopub.execute_input":"2021-06-22T18:43:07.738741Z","iopub.status.idle":"2021-06-22T18:43:07.752068Z","shell.execute_reply.started":"2021-06-22T18:43:07.738715Z","shell.execute_reply":"2021-06-22T18:43:07.751069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Until the end\n\ncounter = 1\n\nfor (test_df, sample_prediction_df) in iter_test:\n    \n    sample_prediction_df_copy = sample_prediction_df.copy()\n    \n    print('------------------------------------ Starting Date No:',counter,' -----------------------------------------')\n    \n    print(sample_prediction_df_copy[0:2])\n    print('-------------------------')\n        \n    sample_prediction_df_copy = sample_prediction_df_copy.reset_index(drop=True)\n    \n    print(sample_prediction_df_copy[0:2])\n    \n    # creat dataset\n    sample_prediction_df_copy['playerId'] = sample_prediction_df_copy['date_playerId']\\\n                                        .map(lambda x: int(x.split('_')[1]))\n    \n    print('Sample Pred Shape :',sample_prediction_df_copy.shape)\n    \n    try:\n        # Example: unpack a dataframe from a json column\n        test_ads_ext = ads_generator(test_df,final_features)\n        test_ads_ext = test_ads_ext.drop_duplicates(subset=['playerId'])\n        print('Pre-Merging of extracted ADS Shape :',test_ads_ext.shape)\n        \n    except:\n        print('Running Alternative Route 1')\n        test_ads_ext = 0\n        \n    \n    players_curr_day = sample_prediction_df_copy[['playerId']].drop_duplicates()\n    print('Pre-Merging of players Shape :',players_curr_day.shape)\n\n    #------------------------------------------------------------------------------------------------\n    if type(test_ads_ext)!=int:\n        \n        try:\n\n            test_ads_ext_final = pd.merge(players_curr_day,test_ads_ext,on=['playerId'],how='left')\n            print('Post Merging final ADS Shape :',test_ads_ext_final.shape)\n\n            test_ads_ext_final = test_ads_ext_final.fillna(0)\n            print('Post Null Treatment ADS Shape :',test_ads_ext_final.shape)\n\n            test_ads_ext_final = test_ads_ext_final[final_features]\n            print('Shape of prediction X :',test_ads_ext_final.shape)\n\n            pred = model.predict(test_ads_ext_final)\n\n            sample_prediction_df['target1'] = np.clip(pred[:,0], 0, 100)\n            sample_prediction_df['target2'] = np.clip(pred[:,1], 0, 100)\n            sample_prediction_df['target3'] = np.clip(pred[:,2], 0, 100)\n            sample_prediction_df['target4'] = np.clip(pred[:,3], 0, 100)\n\n            sample_prediction_df = sample_prediction_df.fillna(0.)\n        \n        except:\n            print('Running Alternative Route 2')\n            # Make your predictions for the next day's engagement\n            sample_prediction_df[\"target1\"] = target_1_med\n            sample_prediction_df[\"target2\"] = target_2_med\n            sample_prediction_df[\"target3\"] = target_3_med\n            sample_prediction_df[\"target4\"] = target_4_med\n            \n        \n    elif test_ads_ext == 0:\n        print('Running Alternative Route 3')\n        # Make your predictions for the next day's engagement\n        sample_prediction_df[\"target1\"] = target_1_med\n        sample_prediction_df[\"target2\"] = target_2_med\n        sample_prediction_df[\"target3\"] = target_3_med\n        sample_prediction_df[\"target4\"] = target_4_med\n          \n    #test_ads_ext_final = test_ads_ext_final.drop_duplicates() #Comment out - suspected\n    \n    #print('Post Merging & dupl drop final ADS Shape :',test_ads_ext_final.shape)\n    \n    \n    # Submit your predictions \n    env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:07.753270Z","iopub.execute_input":"2021-06-22T18:43:07.753702Z","iopub.status.idle":"2021-06-22T18:43:10.939414Z","shell.execute_reply.started":"2021-06-22T18:43:07.753670Z","shell.execute_reply":"2021-06-22T18:43:10.938407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Until the end\n\n# for (test_df, sample_prediction_df) in iter_test:\n    \n#         # Example: unpack a dataframe from a json column\n#         #today_games = unpack_json(test_df['games'].iloc[0])\n    \n#         # Make your predictions for the next day's engagement\n#         sample_prediction_df[\"target1\"] = 0.26764763\n#         sample_prediction_df[\"target2\"] = 0.4\n#         sample_prediction_df[\"target3\"] = 0.5\n#         sample_prediction_df[\"target4\"] = 0.9\n        \n#         sample_prediction_df[\"target1\"] = sample_prediction_df[\"target1\"].astype('float32')\n#         sample_prediction_df[\"target2\"] = sample_prediction_df[\"target2\"].astype('float32')\n#         sample_prediction_df[\"target3\"] = sample_prediction_df[\"target3\"].astype('float32')\n#         sample_prediction_df[\"target4\"] = sample_prediction_df[\"target4\"].astype('float32')\n        \n    \n#         # Submit your predictions \n#         env.predict(sample_prediction_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:10.940901Z","iopub.execute_input":"2021-06-22T18:43:10.941483Z","iopub.status.idle":"2021-06-22T18:43:10.946690Z","shell.execute_reply.started":"2021-06-22T18:43:10.941439Z","shell.execute_reply":"2021-06-22T18:43:10.945256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(sample_prediction_df.dtypes)\n# sample_prediction_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:10.947946Z","iopub.execute_input":"2021-06-22T18:43:10.948249Z","iopub.status.idle":"2021-06-22T18:43:10.961775Z","shell.execute_reply.started":"2021-06-22T18:43:10.948221Z","shell.execute_reply":"2021-06-22T18:43:10.960970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# counter = 1\n\n# #------------------------------------------------------------------------------------------------------------------------\n# for (test_df, sample_prediction_df) in iter_test: # make predictions here\n    \n#     sample_prediction_df_copy = sample_prediction_df.copy()\n    \n#     print('------------------------------------ Starting Date No:',counter,' -----------------------------------------')\n    \n#     print(sample_prediction_df[0:2])\n#     print('-------------------------')\n    \n#     sample_prediction_df = sample_prediction_df.reset_index(drop=True)\n    \n#     print(sample_prediction_df[0:2])\n    \n#     # creat dataset\n#     sample_prediction_df['playerId'] = sample_prediction_df['date_playerId']\\\n#                                         .map(lambda x: int(x.split('_')[1]))\n    \n#     print('Sample Pred Shape :',sample_prediction_df.shape)\n    \n#     test_ads_ext = ads_generator(test_df,final_features)\n#     test_ads_ext = test_ads_ext.drop_duplicates(subset=['playerId'])\n    \n#     players_curr_day = sample_prediction_df[['playerId']].drop_duplicates()\n    \n#     print('Pre-Merging of players Shape :',players_curr_day.shape)\n#     print('Pre-Merging of extracted ADS Shape :',test_ads_ext.shape)\n    \n#     test_ads_ext_final = pd.merge(players_curr_day,test_ads_ext,on=['playerId'],how='left')\n          \n#     print('Post Merging final ADS Shape :',test_ads_ext_final.shape)\n          \n#     test_ads_ext_final = test_ads_ext_final.drop_duplicates()\n    \n#     print('Post Merging & dupl drop final ADS Shape :',test_ads_ext_final.shape)\n    \n#     test_ads_ext_final = test_ads_ext_final.fillna(0)\n          \n#     print('Post Null Treatment ADS Shape :',test_ads_ext_final.shape)\n    \n#     test_ads_ext_final = test_ads_ext_final[final_features]\n#     print('Shape of prediction X :',test_ads_ext_final.shape)\n    \n#     pred = model.predict(test_ads_ext_final)\n    \n#     sample_prediction_df['target1'] = np.clip(pred[:,0], 0, 100)\n#     sample_prediction_df['target2'] = np.clip(pred[:,1], 0, 100)\n#     sample_prediction_df['target3'] = np.clip(pred[:,2], 0, 100)\n#     sample_prediction_df['target4'] = np.clip(pred[:,3], 0, 100)\n    \n#     sample_prediction_df = sample_prediction_df.fillna(0.)\n    \n#     submission = (\n#         sample_prediction_df_copy\n#         [['date_playerId']]\n#         .reset_index()  #  preserve index 'date'\n#         .merge(sample_prediction_df[['date_playerId','target1','target2','target3','target4']],\n#                how='left', on='date_playerId')\n#         .set_index('date')  #  restore index 'date'\n#     )\n    \n#     del test_df\n#     del players_curr_day\n#     del test_ads_ext_final\n#     #sample_prediction_df.drop(columns=['playerId'],inplace=True)\n#     del sample_prediction_df\n#     del sample_prediction_df_copy\n    \n#     print('--------------------------------- Submitting for counter :',counter,'------------------------------------')\n#     counter = counter + 1\n    \n#     #print(sample_prediction_df[0:2])\n    \n#     env.predict(submission)\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:10.963283Z","iopub.execute_input":"2021-06-22T18:43:10.963881Z","iopub.status.idle":"2021-06-22T18:43:10.975074Z","shell.execute_reply.started":"2021-06-22T18:43:10.963835Z","shell.execute_reply":"2021-06-22T18:43:10.974275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(submission.dtypes)\n# submission.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-22T18:43:10.976302Z","iopub.execute_input":"2021-06-22T18:43:10.976875Z","iopub.status.idle":"2021-06-22T18:43:10.991518Z","shell.execute_reply.started":"2021-06-22T18:43:10.976834Z","shell.execute_reply":"2021-06-22T18:43:10.990728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}