{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-04T12:32:16.256644Z","iopub.execute_input":"2021-07-04T12:32:16.257043Z","iopub.status.idle":"2021-07-04T12:32:16.277043Z","shell.execute_reply.started":"2021-07-04T12:32:16.256955Z","shell.execute_reply":"2021-07-04T12:32:16.275584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/train.csv')\ndef unpack_json(json_str):\n    return np.nan if pd.isna(json_str) else pd.read_json(json_str)\ntrain_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T12:32:18.656254Z","iopub.execute_input":"2021-07-04T12:32:18.656748Z","iopub.status.idle":"2021-07-04T12:33:31.15027Z","shell.execute_reply.started":"2021-07-04T12:32:18.656705Z","shell.execute_reply":"2021-07-04T12:33:31.149056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T12:52:05.964209Z","iopub.execute_input":"2021-07-04T12:52:05.964624Z","iopub.status.idle":"2021-07-04T12:52:06.175164Z","shell.execute_reply.started":"2021-07-04T12:52:05.964574Z","shell.execute_reply":"2021-07-04T12:52:06.174445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_1211 = unpack_json(train_df['transactions'].iloc[1211])\nt_1211.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:41:45.739988Z","iopub.execute_input":"2021-07-04T14:41:45.740396Z","iopub.status.idle":"2021-07-04T14:41:45.767779Z","shell.execute_reply.started":"2021-07-04T14:41:45.740358Z","shell.execute_reply":"2021-07-04T14:41:45.765698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:20:28.190373Z","iopub.execute_input":"2021-07-04T13:20:28.190799Z","iopub.status.idle":"2021-07-04T13:20:28.207437Z","shell.execute_reply.started":"2021-07-04T13:20:28.190762Z","shell.execute_reply":"2021-07-04T13:20:28.206219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_1211.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:20:31.935129Z","iopub.execute_input":"2021-07-04T13:20:31.935622Z","iopub.status.idle":"2021-07-04T13:20:31.952134Z","shell.execute_reply.started":"2021-07-04T13:20:31.935573Z","shell.execute_reply":"2021-07-04T13:20:31.950996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"fromTeam NaN은 신인이라 여겨짐,\n\nplayerId를 기준으로 새로운 DF만들기\n\ntransactionId ,playerName,date, fromTeamName, toTemaName, effectiveDate, resolutionDate, typdDesc, description은 삭제\n\ntypeCode는 int형으로 변경하여 사용 <= 어떻게 들어왔는지가 중요할 것 같음 ex)신인?   => 혹여나 아이디어가 부족하면 각 타입별로 0,1 부여하여 corr 해보는 것도 나쁘지는 않을 듯\n\nfromTeamId NaN은 0으로 대체 (팀 코드가 0인 곳은 없다고 전제)","metadata":{}},{"cell_type":"code","source":"target_1211 = unpack_json(train_df['nextDayPlayerEngagement'].iloc[1211])\ntarget_1211.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:19:39.521625Z","iopub.execute_input":"2021-07-04T13:19:39.521975Z","iopub.status.idle":"2021-07-04T13:19:39.557068Z","shell.execute_reply.started":"2021-07-04T13:19:39.521945Z","shell.execute_reply":"2021-07-04T13:19:39.555948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_1211.drop(['transactionId','playerName','date', 'fromTeamName', 'toTeamName','effectiveDate', 'resolutionDate', 'typeDesc', 'description'], axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:20:40.741745Z","iopub.execute_input":"2021-07-04T13:20:40.742104Z","iopub.status.idle":"2021-07-04T13:20:40.750593Z","shell.execute_reply.started":"2021-07-04T13:20:40.742074Z","shell.execute_reply":"2021-07-04T13:20:40.749291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_1211.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:20:47.887548Z","iopub.execute_input":"2021-07-04T13:20:47.88806Z","iopub.status.idle":"2021-07-04T13:20:47.901799Z","shell.execute_reply.started":"2021-07-04T13:20:47.888026Z","shell.execute_reply":"2021-07-04T13:20:47.900569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_1211['fromTeamId'] = t_1211['fromTeamId'].fillna(0)\nt_1211['fromTeamId'] =  t_1211['fromTeamId'].astype(int)\nt_1211.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:30:34.875492Z","iopub.execute_input":"2021-07-04T14:30:34.875865Z","iopub.status.idle":"2021-07-04T14:30:34.892908Z","shell.execute_reply.started":"2021-07-04T14:30:34.875834Z","shell.execute_reply":"2021-07-04T14:30:34.891825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nordinal_encoder = OrdinalEncoder()\nt_1211['typeCode'] = ordinal_encoder.fit_transform(t_1211[['typeCode']])  # 문법상 대괄호가 두개인듯\nt_1211['typeCode'].value_counts()\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:40:06.31382Z","iopub.execute_input":"2021-07-04T13:40:06.314188Z","iopub.status.idle":"2021-07-04T13:40:06.327477Z","shell.execute_reply.started":"2021-07-04T13:40:06.314157Z","shell.execute_reply":"2021-07-04T13:40:06.326143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_1211['typeCode'] = t_1211['typeCode'].astype(int)\nt_1211.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:41:17.285145Z","iopub.execute_input":"2021-07-04T13:41:17.285572Z","iopub.status.idle":"2021-07-04T13:41:17.301071Z","shell.execute_reply.started":"2021-07-04T13:41:17.285537Z","shell.execute_reply":"2021-07-04T13:41:17.299992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"t_1211에 대한 데이터 전처리 완료","metadata":{}},{"cell_type":"code","source":"# merge\ntarget_t = pd.merge(target_1211, t_1211, how = 'outer', on='playerId')","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:48:39.560304Z","iopub.execute_input":"2021-07-04T13:48:39.561053Z","iopub.status.idle":"2021-07-04T13:48:39.584353Z","shell.execute_reply.started":"2021-07-04T13:48:39.560976Z","shell.execute_reply":"2021-07-04T13:48:39.583556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_t = target_t.dropna(subset=['typeCode'])","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:49:42.268127Z","iopub.execute_input":"2021-07-04T13:49:42.268773Z","iopub.status.idle":"2021-07-04T13:49:42.276601Z","shell.execute_reply.started":"2021-07-04T13:49:42.268735Z","shell.execute_reply":"2021-07-04T13:49:42.275626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_t","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:49:46.885985Z","iopub.execute_input":"2021-07-04T13:49:46.886364Z","iopub.status.idle":"2021-07-04T13:49:46.932585Z","shell.execute_reply.started":"2021-07-04T13:49:46.886329Z","shell.execute_reply":"2021-07-04T13:49:46.931593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_t_corr = target_t.corr()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:50:08.322258Z","iopub.execute_input":"2021-07-04T13:50:08.322657Z","iopub.status.idle":"2021-07-04T13:50:08.32796Z","shell.execute_reply.started":"2021-07-04T13:50:08.322624Z","shell.execute_reply":"2021-07-04T13:50:08.326626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_t_corr","metadata":{"execution":{"iopub.status.busy":"2021-07-04T13:50:13.863718Z","iopub.execute_input":"2021-07-04T13:50:13.864225Z","iopub.status.idle":"2021-07-04T13:50:13.879879Z","shell.execute_reply.started":"2021-07-04T13:50:13.864177Z","shell.execute_reply":"2021-07-04T13:50:13.878933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nt_1211 = unpack_json(train_df['transactions'].iloc[1211])\ntarget_1211 = unpack_json(train_df['nextDayPlayerEngagement'].iloc[1211])\nt_1211.drop(['transactionId','playerName','date', 'fromTeamName', 'toTeamName',\n             'effectiveDate', 'resolutionDate', 'typeDesc', 'description'], axis = 1, inplace = True)\nt_1211['fromTeamId'] = t_1211['fromTeamId'].fillna(0)\nt_1211['fromTeamId'] =  t_1211['fromTeamId'].astype(int)\nfrom sklearn.preprocessing import OrdinalEncoder\nordinal_encoder = OrdinalEncoder()\nt_1211['typeCode'] = ordinal_encoder.fit_transform(t_1211[['typeCode']])  # 문법상 대괄호가 두개인듯\nt_1211['typeCode'] = t_1211['typeCode'].astype(int)\ntarget_t = pd.merge(target_1211, t_1211, how = 'outer', on='playerId')\ntarget_t = target_t.dropna(subset=['typeCode'])\ntarget_t_corr = target_t.corr()\n'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nordinal_encoder = OrdinalEncoder()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:27:24.086469Z","iopub.execute_input":"2021-07-04T14:27:24.086996Z","iopub.status.idle":"2021-07-04T14:27:24.091749Z","shell.execute_reply.started":"2021-07-04T14:27:24.086963Z","shell.execute_reply":"2021-07-04T14:27:24.090865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndef transactions_targets_corr(train_df, start, end):\n    plus = lambda a, b : a + b\n    for i in range(start, end):\n        transaction = unpack_json(train_df['transactions'].iloc[i])\n        target = unpack_json(train_df['nextDayPlayerEngagement'].iloc[i])\n        transaction.drop(['transactionId','playerName','date', 'fromTeamName', 'toTeamName',\n             'effectiveDate', 'resolutionDate', 'typeDesc', 'description'], axis = 1, inplace = True)\n        transaction['fromTeamId'] = transaction['fromTeamId'].fillna(0)\n        transaction['fromTeamId'] =  transaction['fromTeamId'].astype(int)\n        transaction['typeCode'] = ordinal_encoder.fit_transform(transaction[['typeCode']])  # 문법상 대괄호가 두개인듯\n        transaction['typeCode'] = transaction['typeCode'].astype(int)\n        trans_target = pd.merge(target, transaction, how = 'outer', on='playerId')\n        trans_target = trans_target.dropna(subset=['typeCode'])\n        trans_target = trans_target.dropna(subset=['typeCode','target1'])\n        trans_target_corr = trans_target.corr()\n        if i == start:\n            t_t_c = trans_target_corr\n        else :\n            t_t_c = t_t_c.combine(trans_target_corr, plus)\n        if i %1 ==0:\n            print(i)\n            print(t_t_c)\n    return t_t_c\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:51:51.92414Z","iopub.execute_input":"2021-07-04T14:51:51.924553Z","iopub.status.idle":"2021-07-04T14:51:51.93454Z","shell.execute_reply.started":"2021-07-04T14:51:51.924518Z","shell.execute_reply":"2021-07-04T14:51:51.933538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df 정제\ntrain_df = pd.read_csv('/kaggle/input/mlb-player-digital-engagement-forecasting/train.csv')\ntrain_df = train_df.dropna(subset=['transactions'])\ntrain_df=train_df.reset_index()\ntrain_df.drop(['index'], axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:32:06.343384Z","iopub.execute_input":"2021-07-04T14:32:06.343804Z","iopub.status.idle":"2021-07-04T14:33:26.022013Z","shell.execute_reply.started":"2021-07-04T14:32:06.343771Z","shell.execute_reply":"2021-07-04T14:33:26.020958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction = unpack_json(train_df['transactions'].iloc[1])\ntarget = unpack_json(train_df['nextDayPlayerEngagement'].iloc[1])\ntransaction.drop(['transactionId','playerName','date', 'fromTeamName', 'toTeamName',\n             'effectiveDate', 'resolutionDate', 'typeDesc', 'description'], axis = 1, inplace = True)\ntransaction['fromTeamId'] = transaction['fromTeamId'].fillna(0)\ntransaction['fromTeamId'] =  transaction['fromTeamId'].astype(int)\ntransaction['typeCode'] = ordinal_encoder.fit_transform(transaction[['typeCode']])  # 문법상 대괄호가 두개인듯\ntransaction['typeCode'] = transaction['typeCode'].astype(int)\ntrans_target = pd.merge(target, transaction, how = 'outer', on='playerId')\ntrans_target = trans_target.dropna(subset=['typeCode'])\ntrans_target = trans_target.dropna(subset=['typeCode','target1'])\ntrans_target","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:53:04.254038Z","iopub.execute_input":"2021-07-04T14:53:04.254595Z","iopub.status.idle":"2021-07-04T14:53:04.314894Z","shell.execute_reply.started":"2021-07-04T14:53:04.254558Z","shell.execute_reply":"2021-07-04T14:53:04.314058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:33:31.053464Z","iopub.execute_input":"2021-07-04T14:33:31.053901Z","iopub.status.idle":"2021-07-04T14:33:31.072181Z","shell.execute_reply.started":"2021-07-04T14:33:31.053844Z","shell.execute_reply":"2021-07-04T14:33:31.070677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_t_c","metadata":{"execution":{"iopub.status.busy":"2021-07-04T14:45:11.783015Z","iopub.execute_input":"2021-07-04T14:45:11.783434Z","iopub.status.idle":"2021-07-04T14:45:11.800519Z","shell.execute_reply.started":"2021-07-04T14:45:11.783382Z","shell.execute_reply":"2021-07-04T14:45:11.799379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"iloc[1] 일때 행값이 1이면 연관도가 책정이 안된다. => 행값이 어떤 기준값이하이면 그 날짜 자체를 버리는게 좋을 듯 => 그 기준은 어떻게 정해야 하나요","metadata":{}}]}