{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-01T12:32:24.723046Z","iopub.execute_input":"2021-07-01T12:32:24.723443Z","iopub.status.idle":"2021-07-01T12:32:24.741198Z","shell.execute_reply.started":"2021-07-01T12:32:24.723358Z","shell.execute_reply":"2021-07-01T12:32:24.740152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Kaggle Challenge\n#Suppress the warning messages\nimport warnings\nwarnings.filterwarnings('ignore')\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\ntrain_data=pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/train.csv',encoding='latin')\ntest_data=pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/example_test.csv',encoding='latin')","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:32:24.742910Z","iopub.execute_input":"2021-07-01T12:32:24.743184Z","iopub.status.idle":"2021-07-01T12:33:34.403429Z","shell.execute_reply.started":"2021-07-01T12:32:24.743157Z","shell.execute_reply":"2021-07-01T12:33:34.400775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:34.407262Z","iopub.execute_input":"2021-07-01T12:33:34.407774Z","iopub.status.idle":"2021-07-01T12:33:34.464840Z","shell.execute_reply.started":"2021-07-01T12:33:34.407708Z","shell.execute_reply":"2021-07-01T12:33:34.464194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting the json file to a dataframe\nimport json\nfrom pandas.io.json import json_normalize\njson_string=train_data['nextDayPlayerEngagement']\nprint(json_string)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:34.466009Z","iopub.execute_input":"2021-07-01T12:33:34.466386Z","iopub.status.idle":"2021-07-01T12:33:34.494282Z","shell.execute_reply.started":"2021-07-01T12:33:34.466358Z","shell.execute_reply":"2021-07-01T12:33:34.493554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def json_to_dataframe (dataframe, column):\n    num_rows = len(dataframe)\n    \n    data_list = []\n    for row in tqdm(range(num_rows)):\n        json_data = dataframe.iloc[row][column]\n        if str(json_data) != \"NaN\":\n            data = pd.read_json(json_data)\n            data_list.append(data)\n        \n    all_data = pd.concat(data_list, axis = 0)\n    \n    return all_data","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:34.495316Z","iopub.execute_input":"2021-07-01T12:33:34.495745Z","iopub.status.idle":"2021-07-01T12:33:34.501161Z","shell.execute_reply.started":"2021-07-01T12:33:34.495711Z","shell.execute_reply":"2021-07-01T12:33:34.500245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_engagement_data=json_to_dataframe(train_data,'nextDayPlayerEngagement')\nplayer_engagement_data=pd.DataFrame(player_engagement_data)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:34.502371Z","iopub.execute_input":"2021-07-01T12:33:34.502983Z","iopub.status.idle":"2021-07-01T12:33:54.586052Z","shell.execute_reply.started":"2021-07-01T12:33:34.502937Z","shell.execute_reply":"2021-07-01T12:33:54.585172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_engagement_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:54.589083Z","iopub.execute_input":"2021-07-01T12:33:54.589397Z","iopub.status.idle":"2021-07-01T12:33:54.603703Z","shell.execute_reply.started":"2021-07-01T12:33:54.589367Z","shell.execute_reply":"2021-07-01T12:33:54.602676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame(player_engagement_data)\ndf['engagementMetricsDate'] = pd.to_datetime(df['engagementMetricsDate'])\ndf['new_date'] = df['engagementMetricsDate'].astype('int64')\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:54.605444Z","iopub.execute_input":"2021-07-01T12:33:54.605766Z","iopub.status.idle":"2021-07-01T12:33:55.566970Z","shell.execute_reply.started":"2021-07-01T12:33:54.605737Z","shell.execute_reply":"2021-07-01T12:33:55.565898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:55.568298Z","iopub.execute_input":"2021-07-01T12:33:55.568881Z","iopub.status.idle":"2021-07-01T12:33:55.583189Z","shell.execute_reply.started":"2021-07-01T12:33:55.568837Z","shell.execute_reply":"2021-07-01T12:33:55.582259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['new_date']=df['new_date']/100000000000","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:55.584615Z","iopub.execute_input":"2021-07-01T12:33:55.584917Z","iopub.status.idle":"2021-07-01T12:33:55.614957Z","shell.execute_reply.started":"2021-07-01T12:33:55.584889Z","shell.execute_reply":"2021-07-01T12:33:55.614057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:55.616011Z","iopub.execute_input":"2021-07-01T12:33:55.616416Z","iopub.status.idle":"2021-07-01T12:33:55.630729Z","shell.execute_reply.started":"2021-07-01T12:33:55.616385Z","shell.execute_reply":"2021-07-01T12:33:55.629548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['date_playerId']=df.new_date.astype(str)+'_'+df.playerId.astype(str)\nnew_dataframe=df","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:33:55.632121Z","iopub.execute_input":"2021-07-01T12:33:55.632438Z","iopub.status.idle":"2021-07-01T12:34:00.119033Z","shell.execute_reply.started":"2021-07-01T12:33:55.632404Z","shell.execute_reply":"2021-07-01T12:34:00.117980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_dataframe.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:34:00.120553Z","iopub.execute_input":"2021-07-01T12:34:00.120860Z","iopub.status.idle":"2021-07-01T12:34:00.137859Z","shell.execute_reply.started":"2021-07-01T12:34:00.120829Z","shell.execute_reply":"2021-07-01T12:34:00.136820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.multioutput import MultiOutputRegressor\n#Random Forest Regressor for target1, target2, target3, target4\nTargetVariable=['target1','target2','target3','target4']\nPredictors=['new_date','playerId']\nX=new_dataframe[Predictors].values\ny=new_dataframe[TargetVariable].values\n\n# Split the data into training and testing set\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)\n#Scaling the data\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nPredictorScaler=StandardScaler()\nPredictorScalerFit=PredictorScaler.fit(X)\nX=PredictorScalerFit.transform(X)\n\nfrom sklearn.ensemble import RandomForestRegressor\nclf =MultiOutputRegressor( RandomForestRegressor(max_depth=6,n_estimators=200,criterion='mse',oob_score=True))\nprint(clf)\nRF=clf.fit(X_train,y_train)\nprediction=RF.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T12:47:48.412179Z","iopub.execute_input":"2021-07-01T12:47:48.412632Z","iopub.status.idle":"2021-07-01T13:19:26.774218Z","shell.execute_reply.started":"2021-07-01T12:47:48.412591Z","shell.execute_reply":"2021-07-01T13:19:26.773247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def my_custom_loss_func(y_test, prediction):\n    diff = np.abs(y_test - prediction).max()\n    return np.log1p(diff)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:22:40.447383Z","iopub.execute_input":"2021-07-01T13:22:40.447802Z","iopub.status.idle":"2021-07-01T13:22:40.452124Z","shell.execute_reply.started":"2021-07-01T13:22:40.447758Z","shell.execute_reply":"2021-07-01T13:22:40.451487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_custom_loss_func(y_test,prediction)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:22:51.067444Z","iopub.execute_input":"2021-07-01T13:22:51.067843Z","iopub.status.idle":"2021-07-01T13:22:51.091292Z","shell.execute_reply.started":"2021-07-01T13:22:51.067811Z","shell.execute_reply":"2021-07-01T13:22:51.090528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import fbeta_score, make_scorer\nscore = make_scorer(my_custom_loss_func, greater_is_better=False)\nscore(clf,X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:22:57.458914Z","iopub.execute_input":"2021-07-01T13:22:57.459257Z","iopub.status.idle":"2021-07-01T13:24:07.063910Z","shell.execute_reply.started":"2021-07-01T13:22:57.459226Z","shell.execute_reply":"2021-07-01T13:24:07.062695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\n\nprint('Mean Absolute Error:', metrics.mean_absolute_error(y_test,prediction))\nprint('Mean Squared Error:', metrics.mean_squared_error(y_test, prediction))\nprint('Root Mean Squared Error:', np.sqrt(metrics.mean_squared_error(y_test, prediction)))","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:24:24.281184Z","iopub.execute_input":"2021-07-01T13:24:24.281711Z","iopub.status.idle":"2021-07-01T13:24:24.350779Z","shell.execute_reply.started":"2021-07-01T13:24:24.281674Z","shell.execute_reply":"2021-07-01T13:24:24.349543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=pd.DataFrame(prediction)\nprediction.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:25:04.034622Z","iopub.execute_input":"2021-07-01T13:25:04.034998Z","iopub.status.idle":"2021-07-01T13:25:04.047888Z","shell.execute_reply.started":"2021-07-01T13:25:04.034965Z","shell.execute_reply":"2021-07-01T13:25:04.046448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.columns=['targetpred1','targetpred2','targetpred3','targetpred4']\nprediction.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:25:12.590281Z","iopub.execute_input":"2021-07-01T13:25:12.590667Z","iopub.status.idle":"2021-07-01T13:25:12.602405Z","shell.execute_reply.started":"2021-07-01T13:25:12.590631Z","shell.execute_reply":"2021-07-01T13:25:12.601299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test=pd.DataFrame(X_test)\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:25:34.129489Z","iopub.execute_input":"2021-07-01T13:25:34.129851Z","iopub.status.idle":"2021-07-01T13:25:34.140203Z","shell.execute_reply.started":"2021-07-01T13:25:34.129816Z","shell.execute_reply":"2021-07-01T13:25:34.139013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.columns=['new_date','playerId']\nX_test['date_playerId']=X_test.new_date.astype(str)+'_'+X_test.playerId.astype(str)\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:26:58.752715Z","iopub.execute_input":"2021-07-01T13:26:58.753244Z","iopub.status.idle":"2021-07-01T13:26:59.548473Z","shell.execute_reply.started":"2021-07-01T13:26:58.753210Z","shell.execute_reply":"2021-07-01T13:26:59.547641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train=pd.DataFrame(X_train)\nX_train.head()\n#X_train.columns=['new_date','playerId']\n#X_train['date_playerId']=X_train.new_date.astype(str)+'_'+X_train.playerId.astype(str)\n#X_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:39:23.032343Z","iopub.execute_input":"2021-07-01T13:39:23.032849Z","iopub.status.idle":"2021-07-01T13:39:23.045511Z","shell.execute_reply.started":"2021-07-01T13:39:23.032807Z","shell.execute_reply":"2021-07-01T13:39:23.044485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_=X_test[['new_date','date_playerId']]\noutput_.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:30:35.455506Z","iopub.execute_input":"2021-07-01T13:30:35.456073Z","iopub.status.idle":"2021-07-01T13:30:35.478607Z","shell.execute_reply.started":"2021-07-01T13:30:35.456022Z","shell.execute_reply":"2021-07-01T13:30:35.477766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_['target1_pred']=prediction.targetpred1.astype(str)\noutput_['target2_pred']=prediction.targetpred2.astype(str)\noutput_['target3_pred']=prediction.targetpred3.astype(str)\noutput_['target4_pred']=prediction.targetpred4.astype(str)\noutput_.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:30:47.646833Z","iopub.execute_input":"2021-07-01T13:30:47.647172Z","iopub.status.idle":"2021-07-01T13:30:49.801226Z","shell.execute_reply.started":"2021-07-01T13:30:47.647143Z","shell.execute_reply":"2021-07-01T13:30:49.800356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_.to_csv(r'submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-01T13:30:56.897195Z","iopub.execute_input":"2021-07-01T13:30:56.897714Z","iopub.status.idle":"2021-07-01T13:30:59.104801Z","shell.execute_reply.started":"2021-07-01T13:30:56.897681Z","shell.execute_reply":"2021-07-01T13:30:59.103905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}