{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n!pip install feature_engine 2>/dev/null 1>&2\n!pip install fastparquet 2>/dev/null 1>&2\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\n\nfrom sklearn.preprocessing import StandardScaler\nfrom feature_engine.wrappers import SklearnTransformerWrapper as SKWrapper\nfrom sklearn.model_selection import train_test_split\n\n\nfrom matplotlib import pyplot as plt\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-28T05:11:40.826361Z","iopub.execute_input":"2022-10-28T05:11:40.826836Z","iopub.status.idle":"2022-10-28T05:12:02.906999Z","shell.execute_reply.started":"2022-10-28T05:11:40.826800Z","shell.execute_reply":"2022-10-28T05:12:02.905370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Downloading data**","metadata":{}},{"cell_type":"code","source":"INPUT = '../input/tabular-playground-series-oct-2022/'\n\ndf_train_dtypes = pd.read_csv(INPUT + 'train_dtypes.csv')\ndf_test_dtypes = pd.read_csv(INPUT + 'test_dtypes.csv')\ntrain_dtypes = {k: v for (k, v) in zip(df_train_dtypes.column, df_train_dtypes.dtype)}\ntest_dtypes = {k: v for (k, v) in zip(df_test_dtypes.column, df_test_dtypes.dtype)}\n\nall_columns =list(pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/train_0.csv\",nrows=1))\ndropcols = ['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next']\nboost_t = ['boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer', 'boost4_timer', 'boost5_timer']\ndropcols = dropcols + boost_t\nusecols = [i for i in all_columns if i not in dropcols]","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:12:02.910050Z","iopub.execute_input":"2022-10-28T05:12:02.910582Z","iopub.status.idle":"2022-10-28T05:12:02.947633Z","shell.execute_reply.started":"2022-10-28T05:12:02.910530Z","shell.execute_reply":"2022-10-28T05:12:02.946407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = []\nnum = 4\nt_list = [3, 5, 7, 9]\nfor i in t_list:\n#     df = pd.read_csv(INPUT + f'train_{i}.csv', dtype = train_dtypes)\n    df = pd.read_csv(INPUT + f'train_{i}.csv', usecols = usecols)\n    df.to_parquet(f'train_{i}.parquet.gzip', compression='gzip')\n    print('Done with File', i)\n    train.append(pd.read_parquet(f'train_{i}.parquet.gzip'))\n\n# dft = pd.read_csv(INPUT + 'test.csv', dtype = test_dtypes)\ndft = pd.read_csv(INPUT + 'test.csv')\ndft.to_parquet('test.parquet.gzip', compression='gzip')\nprint('Done with File test')\ntest = pd.read_parquet('test.parquet.gzip')\ndf_sample = pd.read_csv(INPUT + 'sample_submission.csv')\n\ntest = test.drop(boost_t, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:12:02.949441Z","iopub.execute_input":"2022-10-28T05:12:02.949870Z","iopub.status.idle":"2022-10-28T05:16:56.408680Z","shell.execute_reply.started":"2022-10-28T05:12:02.949834Z","shell.execute_reply":"2022-10-28T05:16:56.407518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Preprocessing data**\n\nThanks to @Jose Cáliz for feature engineering ideas!","metadata":{}},{"cell_type":"code","source":"# for i in range(num):\n#     print(train_list[i].shape)\n#     games = random.sample(list(train_list[i].game_num.unique()), 300)\n#     train_list[i] = train_list[i][train_list[i].game_num.isin(games)]\n#     print(train_list[i].shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:16:56.411548Z","iopub.execute_input":"2022-10-28T05:16:56.411916Z","iopub.status.idle":"2022-10-28T05:16:56.417251Z","shell.execute_reply.started":"2022-10-28T05:16:56.411882Z","shell.execute_reply":"2022-10-28T05:16:56.416035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(num):\n    train[i]['label'] = train[i].team_A_scoring_within_10sec + train[i].team_B_scoring_within_10sec.replace(1, 2)\n    train[i].label.value_counts(True).to_frame(name='label proportion')","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:16:56.419026Z","iopub.execute_input":"2022-10-28T05:16:56.419497Z","iopub.status.idle":"2022-10-28T05:16:56.579822Z","shell.execute_reply.started":"2022-10-28T05:16:56.419450Z","shell.execute_reply":"2022-10-28T05:16:56.578473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(num):\n#     train[i] = train[i].fillna(train[i].median())\n    train[i] = train[i].fillna(0)\n    \n# test = test.fillna(test.median())\ntest = test.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:16:56.581076Z","iopub.execute_input":"2022-10-28T05:16:56.581410Z","iopub.status.idle":"2022-10-28T05:16:59.289747Z","shell.execute_reply.started":"2022-10-28T05:16:56.581380Z","shell.execute_reply":"2022-10-28T05:16:59.288207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Feature engineering**","metadata":{}},{"cell_type":"code","source":"for i in range(num):\n    train[i]['ball_distance_to_goal_A'] = np.sqrt(\n        (train[i].ball_pos_x)**2 + (train[i].ball_pos_y + 100)**2 + (train[i].ball_pos_z)**2\n    )\n    train[i]['ball_distance_to_goal_B'] = np.sqrt(\n        (train[i].ball_pos_x)**2 + (train[i].ball_pos_y - 100)**2 + (train[i].ball_pos_z)**2\n    )\n    \n    train[i]['ball_pos'] = np.sqrt(train[i]['ball_pos_x']**2 + train[i]['ball_pos_y']**2 + train[i]['ball_pos_z']**2)\n    train[i]['ball_vel'] = np.sqrt(train[i]['ball_vel_x']**2 + train[i]['ball_vel_y']**2 + train[i]['ball_vel_z']**2)\n    \n    train[i]['p0_pos'] = np.sqrt(train[i]['p0_pos_x']**2 + train[i]['p0_pos_y']**2 + train[i]['p0_pos_z']**2)\n    train[i]['p1_pos'] = np.sqrt(train[i]['p1_pos_x']**2 + train[i]['p1_pos_y']**2 + train[i]['p1_pos_z']**2)\n    train[i]['p2_pos'] = np.sqrt(train[i]['p2_pos_x']**2 + train[i]['p2_pos_y']**2 + train[i]['p2_pos_z']**2)\n    train[i]['p3_pos'] = np.sqrt(train[i]['p3_pos_x']**2 + train[i]['p3_pos_y']**2 + train[i]['p3_pos_z']**2)\n    train[i]['p4_pos'] = np.sqrt(train[i]['p4_pos_x']**2 + train[i]['p4_pos_y']**2 + train[i]['p4_pos_z']**2)\n    train[i]['p5_pos'] = np.sqrt(train[i]['p5_pos_x']**2 + train[i]['p5_pos_y']**2 + train[i]['p5_pos_z']**2)\n    \n    train[i]['p0_vel'] = np.sqrt(train[i]['p0_vel_x']**2 + train[i]['p0_vel_y']**2 + train[i]['p0_vel_z']**2)\n    train[i]['p1_vel'] = np.sqrt(train[i]['p1_vel_x']**2 + train[i]['p1_vel_y']**2 + train[i]['p1_vel_z']**2)\n    train[i]['p2_vel'] = np.sqrt(train[i]['p2_vel_x']**2 + train[i]['p2_vel_y']**2 + train[i]['p2_vel_z']**2)\n    train[i]['p3_vel'] = np.sqrt(train[i]['p3_vel_x']**2 + train[i]['p3_vel_y']**2 + train[i]['p3_vel_z']**2)\n    train[i]['p4_vel'] = np.sqrt(train[i]['p4_vel_x']**2 + train[i]['p4_vel_y']**2 + train[i]['p4_vel_z']**2)\n    train[i]['p5_vel'] = np.sqrt(train[i]['p5_vel_x']**2 + train[i]['p5_vel_y']**2 + train[i]['p5_vel_z']**2)\n    \n    train[i]['p0_to_ball'] = np.sqrt((train[i]['p0_pos_x']-train[i]['ball_pos_x'])**2 + (train[i]['p0_pos_y'] - train[i]['ball_pos_y'])**2 + (train[i]['p0_pos_z'] - train[i]['ball_pos_z'])**2)\n    train[i]['p1_to_ball'] = np.sqrt((train[i]['p1_pos_x']-train[i]['ball_pos_x'])**2 + (train[i]['p1_pos_y'] - train[i]['ball_pos_y'])**2 + (train[i]['p1_pos_z'] - train[i]['ball_pos_z'])**2)\n    train[i]['p2_to_ball'] = np.sqrt((train[i]['p2_pos_x']-train[i]['ball_pos_x'])**2 + (train[i]['p2_pos_y'] - train[i]['ball_pos_y'])**2 + (train[i]['p2_pos_z'] - train[i]['ball_pos_z'])**2)\n    train[i]['p3_to_ball'] = np.sqrt((train[i]['p3_pos_x']-train[i]['ball_pos_x'])**2 + (train[i]['p3_pos_y'] - train[i]['ball_pos_y'])**2 + (train[i]['p3_pos_z'] - train[i]['ball_pos_z'])**2)\n    train[i]['p4_to_ball'] = np.sqrt((train[i]['p4_pos_x']-train[i]['ball_pos_x'])**2 + (train[i]['p4_pos_y'] - train[i]['ball_pos_y'])**2 + (train[i]['p4_pos_z'] - train[i]['ball_pos_z'])**2)\n    train[i]['p5_to_ball'] = np.sqrt((train[i]['p5_pos_x']-train[i]['ball_pos_x'])**2 + (train[i]['p5_pos_y'] - train[i]['ball_pos_y'])**2 + (train[i]['p5_pos_z'] - train[i]['ball_pos_z'])**2)\n    \n    train[i]['p0_to_goal'] = np.sqrt((train[i]['p0_pos_x'])**2 + (train[i]['p0_pos_y'] + 100)**2 + (train[i]['p0_pos_z'])**2)\n    train[i]['p1_to_goal'] = np.sqrt((train[i]['p1_pos_x'])**2 + (train[i]['p1_pos_y'] + 100)**2 + (train[i]['p1_pos_z'])**2)\n    train[i]['p2_to_goal'] = np.sqrt((train[i]['p2_pos_x'])**2 + (train[i]['p2_pos_y'] + 100)**2 + (train[i]['p2_pos_z'])**2)\n    train[i]['p3_to_goal'] = np.sqrt((train[i]['p3_pos_x'])**2 + (train[i]['p3_pos_y'] - 100)**2 + (train[i]['p3_pos_z'])**2)\n    train[i]['p4_to_goal'] = np.sqrt((train[i]['p4_pos_x'])**2 + (train[i]['p4_pos_y'] - 100)**2 + (train[i]['p4_pos_z'])**2)\n    train[i]['p5_to_goal'] = np.sqrt((train[i]['p5_pos_x'])**2 + (train[i]['p5_pos_y'] - 100)**2 + (train[i]['p5_pos_z'])**2)\n    \ntest['ball_distance_to_goal_A'] = np.sqrt(\n        (test.ball_pos_x)**2 + (test.ball_pos_y + 100)**2 + (test.ball_pos_z)**2\n    )\ntest['ball_distance_to_goal_B'] = np.sqrt(\n        (test.ball_pos_x)**2 + (test.ball_pos_y - 100)**2 + (test.ball_pos_z)**2\n    )\n\ntest['ball_pos'] = np.sqrt(test['ball_pos_x']**2 + test['ball_pos_y']**2 + test['ball_pos_z']**2)\ntest['ball_vel'] = np.sqrt(test['ball_vel_x']**2 + test['ball_vel_y']**2 + test['ball_vel_z']**2)\n\ntest['p0_pos'] = np.sqrt(test['p0_pos_x']**2 + test['p0_pos_y']**2 + test['p0_pos_z']**2)\ntest['p1_pos'] = np.sqrt(test['p1_pos_x']**2 + test['p1_pos_y']**2 + test['p1_pos_z']**2)\ntest['p2_pos'] = np.sqrt(test['p2_pos_x']**2 + test['p2_pos_y']**2 + test['p2_pos_z']**2)\ntest['p3_pos'] = np.sqrt(test['p3_pos_x']**2 + test['p3_pos_y']**2 + test['p3_pos_z']**2)\ntest['p4_pos'] = np.sqrt(test['p4_pos_x']**2 + test['p4_pos_y']**2 + test['p4_pos_z']**2)\ntest['p5_pos'] = np.sqrt(test['p5_pos_x']**2 + test['p5_pos_y']**2 + test['p5_pos_z']**2)\n    \ntest['p0_vel'] = np.sqrt(test['p0_vel_x']**2 + test['p0_vel_y']**2 + test['p0_vel_z']**2)\ntest['p1_vel'] = np.sqrt(test['p1_vel_x']**2 + test['p1_vel_y']**2 + test['p1_vel_z']**2)\ntest['p2_vel'] = np.sqrt(test['p2_vel_x']**2 + test['p2_vel_y']**2 + test['p2_vel_z']**2)\ntest['p3_vel'] = np.sqrt(test['p3_vel_x']**2 + test['p3_vel_y']**2 + test['p3_vel_z']**2)\ntest['p4_vel'] = np.sqrt(test['p4_vel_x']**2 + test['p4_vel_y']**2 + test['p4_vel_z']**2)\ntest['p5_vel'] = np.sqrt(test['p5_vel_x']**2 + test['p5_vel_y']**2 + test['p5_vel_z']**2)\n    \ntest['p0_to_ball'] = np.sqrt((test['p0_pos_x'] - test['ball_pos_x'])**2 + (test['p0_pos_y'] - test['ball_pos_y'])**2 + (test['p0_pos_z'] - test['ball_pos_z'])**2)\ntest['p1_to_ball'] = np.sqrt((test['p1_pos_x'] - test['ball_pos_x'])**2 + (test['p1_pos_y'] - test['ball_pos_y'])**2 + (test['p1_pos_z'] - test['ball_pos_z'])**2)\ntest['p2_to_ball'] = np.sqrt((test['p2_pos_x'] - test['ball_pos_x'])**2 + (test['p2_pos_y'] - test['ball_pos_y'])**2 + (test['p2_pos_z'] - test['ball_pos_z'])**2)\ntest['p3_to_ball'] = np.sqrt((test['p3_pos_x'] - test['ball_pos_x'])**2 + (test['p3_pos_y'] - test['ball_pos_y'])**2 + (test['p3_pos_z'] - test['ball_pos_z'])**2)\ntest['p4_to_ball'] = np.sqrt((test['p4_pos_x'] - test['ball_pos_x'])**2 + (test['p4_pos_y'] - test['ball_pos_y'])**2 + (test['p4_pos_z'] - test['ball_pos_z'])**2)\ntest['p5_to_ball'] = np.sqrt((test['p5_pos_x'] - test['ball_pos_x'])**2 + (test['p5_pos_y'] - test['ball_pos_y'])**2 + (test['p5_pos_z'] - test['ball_pos_z'])**2)\n\ntest['p0_to_goal'] = np.sqrt((test['p0_pos_x'])**2 + (test['p0_pos_y'] + 100)**2 + (test['p0_pos_z'])**2)\ntest['p1_to_goal'] = np.sqrt((test['p1_pos_x'])**2 + (test['p1_pos_y'] + 100)**2 + (test['p1_pos_z'])**2)\ntest['p2_to_goal'] = np.sqrt((test['p2_pos_x'])**2 + (test['p2_pos_y'] + 100)**2 + (test['p2_pos_z'])**2)\ntest['p3_to_goal'] = np.sqrt((test['p3_pos_x'])**2 + (test['p3_pos_y'] - 100)**2 + (test['p3_pos_z'])**2)\ntest['p4_to_goal'] = np.sqrt((test['p4_pos_x'])**2 + (test['p4_pos_y'] - 100)**2 + (test['p4_pos_z'])**2)\ntest['p5_to_goal'] = np.sqrt((test['p5_pos_x'])**2 + (test['p5_pos_y'] - 100)**2 + (test['p5_pos_z'])**2)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:16:59.291470Z","iopub.execute_input":"2022-10-28T05:16:59.291788Z","iopub.status.idle":"2022-10-28T05:17:09.451183Z","shell.execute_reply.started":"2022-10-28T05:16:59.291760Z","shell.execute_reply":"2022-10-28T05:17:09.450024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.decomposition import PCA\n\n# ball_transform = PCA(4)\n\n# ball_cols = ['ball_vel_x', 'ball_vel_y', 'ball_vel_z', 'ball_pos_x', 'ball_pos_y', 'ball_pos_z']\n\n# ball_transform.fit(train[0][ball_cols])\n\n# for i in range(num):\n#     x = ball_transform.transform(train[i][ball_cols])\n#     train[i] = train[i].drop(ball_cols, axis=1)\n#     x = x.T\n#     train[i]['ball0'], train[i]['ball1'], train[i]['ball2'], train[i]['ball3'] = x[0], x[1], x[2], x[3]\n    \n# x = ball_transform.transform(test[ball_cols])\n# test = test.drop(ball_cols, axis=1)\n# x = x.T\n# test['ball0'], test['ball1'], test['ball2'], test['ball3'] = x[0], x[1], x[2], x[3]\n\n# ball_transform.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.452638Z","iopub.execute_input":"2022-10-28T05:17:09.453515Z","iopub.status.idle":"2022-10-28T05:17:09.457675Z","shell.execute_reply.started":"2022-10-28T05:17:09.453479Z","shell.execute_reply":"2022-10-28T05:17:09.456725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pl0_cols = ['p0_pos_x', 'p0_pos_y', 'p0_pos_z']\n# pl1_cols = ['p1_pos_x', 'p1_pos_y', 'p1_pos_z']\n# pl2_cols = ['p2_pos_x', 'p2_pos_y', 'p2_pos_z']\n# pl3_cols = ['p3_pos_x', 'p3_pos_y', 'p3_pos_z']\n# pl4_cols = ['p4_pos_x', 'p4_pos_y', 'p4_pos_z']\n# pl5_cols = ['p5_pos_x', 'p5_pos_y', 'p5_pos_z']\n\n# pl0_v_cols = ['p0_vel_x', 'p0_vel_y', 'p0_vel_z']\n# pl1_v_cols = ['p1_vel_x', 'p1_vel_y', 'p1_vel_z']\n# pl2_v_cols = ['p2_vel_x', 'p2_vel_y', 'p2_vel_z']\n# pl3_v_cols = ['p3_vel_x', 'p3_vel_y', 'p3_vel_z']\n# pl4_v_cols = ['p4_vel_x', 'p4_vel_y', 'p4_vel_z']\n# pl5_v_cols = ['p5_vel_x', 'p5_vel_y', 'p5_vel_z']\n\n# pl_v_z_cols = ['p0_vel_z', 'p1_vel_z', 'p2_vel_z', 'p3_vel_z', 'p4_vel_z', 'p5_vel_z']\n\n# pl = pl0_cols + pl1_cols + pl2_cols + pl3_cols + pl4_cols + pl5_cols\n# pl_v = pl0_v_cols + pl1_v_cols + pl2_v_cols + pl3_v_cols + pl4_v_cols + pl5_v_cols\n\n# players_transform = PCA(2)\n\n# players_transform.fit(train[0][pl])\n\n# for i in range(num):\n#     x = players_transform.transform(train[i][pl])\n#     x = x.T\n#     train[i] = train[i].drop(pl, axis=1)\n#     train[i]['pl0'], train[i]['pl1'] = x[0], x[1]\n    \n# x = players_transform.transform(test[pl])\n# x = x.T\n# test = test.drop(pl, axis=1)\n# test['pl0'], test['pl1'] = x[0], x[1]\n\n# for i in range(num):\n#     train[i] = train[i].drop(pl_v_z_cols, axis=1)\n# test = test.drop(pl_v_z_cols, axis=1)\n\n# players_transform.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.458691Z","iopub.execute_input":"2022-10-28T05:17:09.459745Z","iopub.status.idle":"2022-10-28T05:17:09.470131Z","shell.execute_reply.started":"2022-10-28T05:17:09.459709Z","shell.execute_reply":"2022-10-28T05:17:09.468859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pl_boost = ['p0_boost', 'p1_boost', 'p2_boost', 'p3_boost', 'p4_boost', 'p5_boost']\n\n# pl_boost_transform = PCA(6)\n\n# pl_boost_transform.fit(train[0][pl_boost])\n\n# for i in range(num):\n#     x = pl_boost_transform.transform(train[i][pl_boost])\n#     x = x.T\n#     train[i] = train[i].drop(pl_boost, axis=1)\n#     train[i]['pl_b1'], train[i]['pl_b2'] = x[0], x[1]\n    \n# x = pl_boost_transform.transform(test[pl_boost])\n# x = x.T\n# test = test.drop(pl_boost, axis=1)\n# test['pl_b1'], test['pl_b2'] = x[0], x[1]\n\n# pl_boost_transform.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.473937Z","iopub.execute_input":"2022-10-28T05:17:09.474298Z","iopub.status.idle":"2022-10-28T05:17:09.484442Z","shell.execute_reply.started":"2022-10-28T05:17:09.474267Z","shell.execute_reply":"2022-10-28T05:17:09.483387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# boost_t = ['boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer', 'boost4_timer', 'boost5_timer']\n\n# boost_t_transform = PCA(6)\n\n# boost_t_transform.fit(train[0][boost_t])\n\n# for i in range(num):\n#     x = boost_t_transform.transform(train[i][boost_t])\n#     x = x.T\n#     train[i] = train[i].drop(boost_t, axis=1)\n#     train[i]['b1'], train[i]['b2'] = x[0], x[1]\n    \n# x = boost_t_transform.transform(test[boost_t])\n# x = x.T\n# test = test.drop(boost_t, axis=1)\n# test['b1'], test['b2'] = x[0], x[1]\n\n# boost_t_transform.explained_variance_ratio_\n\n# for i in range(num):\n#     train[i] = train[i].drop(boost_t, axis=1)\n# test = test.drop(boost_t, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.485961Z","iopub.execute_input":"2022-10-28T05:17:09.486336Z","iopub.status.idle":"2022-10-28T05:17:09.495101Z","shell.execute_reply.started":"2022-10-28T05:17:09.486302Z","shell.execute_reply":"2022-10-28T05:17:09.494054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.496188Z","iopub.execute_input":"2022-10-28T05:17:09.496533Z","iopub.status.idle":"2022-10-28T05:17:09.814013Z","shell.execute_reply.started":"2022-10-28T05:17:09.496504Z","shell.execute_reply":"2022-10-28T05:17:09.812881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mirror_x(df, columns):\n    data = pd.DataFrame(df, columns = columns)\n    data['ball_pos_x'] = - data['ball_pos_x']\n    data['ball_vel_x'] = - data['ball_vel_x']\n    data['p0_pos_x'] = - data['p0_pos_x']\n    data['p0_vel_x'] = - data['p0_vel_x']\n    data['p1_pos_x'] = - data['p1_pos_x']\n    data['p1_vel_x'] = - data['p1_vel_x']\n    data['p2_pos_x'] = - data['p2_pos_x']\n    data['p2_vel_x'] = - data['p2_vel_x']\n    data['p3_pos_x'] = - data['p3_pos_x']\n    data['p3_vel_x'] = - data['p3_vel_x']\n    data['p4_pos_x'] = - data['p4_pos_x']\n    data['p4_vel_x'] = - data['p4_vel_x']\n    data['p5_pos_x'] = - data['p5_pos_x']\n    data['p5_vel_x'] = - data['p5_vel_x']\n    return data.to_numpy() ","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.815773Z","iopub.execute_input":"2022-10-28T05:17:09.816245Z","iopub.status.idle":"2022-10-28T05:17:09.826665Z","shell.execute_reply.started":"2022-10-28T05:17:09.816180Z","shell.execute_reply":"2022-10-28T05:17:09.825221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def change_players(df, columns):\n    data = pd.DataFrame(df, columns = columns)\n    data['p0_pos_x'], data['p1_pos_x'], data['p2_pos_x'] = data['p1_pos_x'], data['p2_pos_x'], data['p0_pos_x']\n    data['p0_pos_y'], data['p1_pos_y'], data['p2_pos_y'] = data['p1_pos_y'], data['p2_pos_y'], data['p0_pos_y']\n    data['p0_pos_z'], data['p1_pos_z'], data['p2_pos_z'] = data['p1_pos_z'], data['p2_pos_z'], data['p0_pos_z']\n    data['p0_vel_x'], data['p1_vel_x'], data['p2_vel_x'] = data['p1_vel_x'], data['p2_vel_x'], data['p0_vel_x']\n    data['p0_vel_y'], data['p1_vel_y'], data['p2_vel_y'] = data['p1_vel_y'], data['p2_vel_y'], data['p0_vel_y']\n    data['p0_vel_z'], data['p1_vel_z'], data['p2_vel_z'] = data['p1_vel_z'], data['p2_vel_z'], data['p0_vel_z']\n    data['p0_boost'], data['p1_boost'], data['p2_boost'] = data['p1_boost'], data['p2_boost'], data['p0_boost']\n    \n    data['p3_pos_x'], data['p4_pos_x'], data['p5_pos_x'] = data['p4_pos_x'], data['p5_pos_x'], data['p3_pos_x']\n    data['p3_pos_y'], data['p4_pos_y'], data['p5_pos_y'] = data['p4_pos_y'], data['p5_pos_y'], data['p3_pos_y']\n    data['p3_pos_z'], data['p4_pos_z'], data['p5_pos_z'] = data['p4_pos_z'], data['p5_pos_z'], data['p3_pos_z']\n    data['p3_vel_x'], data['p4_vel_x'], data['p5_vel_x'] = data['p4_vel_x'], data['p5_vel_x'], data['p3_vel_x']\n    data['p3_vel_y'], data['p4_vel_y'], data['p5_vel_y'] = data['p4_vel_y'], data['p5_vel_y'], data['p3_vel_y']\n    data['p3_vel_z'], data['p4_vel_z'], data['p5_vel_z'] = data['p4_vel_z'], data['p5_vel_z'], data['p3_vel_z']\n    data['p3_boost'], data['p4_boost'], data['p5_boost'] = data['p4_boost'], data['p5_boost'], data['p3_boost']\n    return data.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.828359Z","iopub.execute_input":"2022-10-28T05:17:09.828718Z","iopub.status.idle":"2022-10-28T05:17:09.843107Z","shell.execute_reply.started":"2022-10-28T05:17:09.828685Z","shell.execute_reply":"2022-10-28T05:17:09.841716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_inv(df, columns):\n    data = pd.DataFrame(df, columns = columns)\n    data['ball_distance_to_goal_A'], data['ball_distance_to_goal_B'] = data['ball_distance_to_goal_B'], data['ball_distance_to_goal_A']\n    data['ball_pos_y'] = - data['ball_pos_y']\n    data['ball_vel_y'] = - data['ball_vel_y']\n    data['p0_pos_x'], data['p3_pos_x'] = - data['p3_pos_x'], - data['p0_pos_x']\n    data['p0_pos_y'], data['p3_pos_y'] = - data['p3_pos_y'], - data['p0_pos_y']\n    data['p0_pos_z'], data['p3_pos_z'] = data['p3_pos_z'], data['p0_pos_z']\n    data['p0_vel_x'], data['p3_vel_x'] = - data['p3_vel_x'], - data['p0_vel_x']\n    data['p0_vel_y'], data['p3_vel_y'] = - data['p3_vel_y'], - data['p0_vel_y']\n    data['p0_vel_z'], data['p3_vel_z'] = data['p3_vel_z'], data['p0_vel_z']\n    data['p1_pos_x'], data['p4_pos_x'] = - data['p4_pos_x'], - data['p1_pos_x']\n    data['p1_pos_y'], data['p4_pos_y'] = - data['p4_pos_y'], - data['p1_pos_y']\n    data['p1_pos_z'], data['p4_pos_z'] = data['p4_pos_z'], data['p1_pos_z']\n    data['p1_vel_x'], data['p4_vel_x'] = - data['p4_vel_x'], - data['p1_vel_x']\n    data['p1_vel_y'], data['p4_vel_y'] = - data['p4_vel_y'], - data['p1_vel_y']\n    data['p1_vel_z'], data['p4_vel_z'] = data['p4_vel_z'], data['p1_vel_z']\n    data['p2_pos_x'], data['p5_pos_x'] = - data['p5_pos_x'], - data['p2_pos_x']\n    data['p2_pos_y'], data['p5_pos_y'] = - data['p5_pos_y'], - data['p2_pos_y']\n    data['p2_pos_z'], data['p5_pos_z'] = data['p5_pos_z'], data['p2_pos_z']\n    data['p2_vel_x'], data['p5_vel_x'] = - data['p5_vel_x'], - data['p2_vel_x']\n    data['p2_vel_y'], data['p5_vel_y'] = - data['p5_vel_y'], - data['p2_vel_y']\n    data['p2_vel_z'], data['p5_vel_z'] = data['p5_vel_z'], data['p2_vel_z']\n    data['p0_boost'], data['p3_boost'] = data['p3_boost'], data['p0_boost']\n    data['p1_boost'], data['p4_boost'] = data['p4_boost'], data['p1_boost']\n    data['p2_boost'], data['p5_boost'] = data['p5_boost'], data['p2_boost']\n    return data.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.844814Z","iopub.execute_input":"2022-10-28T05:17:09.845212Z","iopub.status.idle":"2022-10-28T05:17:09.861404Z","shell.execute_reply.started":"2022-10-28T05:17:09.845160Z","shell.execute_reply":"2022-10-28T05:17:09.860293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:09.863056Z","iopub.execute_input":"2022-10-28T05:17:09.863424Z","iopub.status.idle":"2022-10-28T05:17:10.143386Z","shell.execute_reply.started":"2022-10-28T05:17:09.863393Z","shell.execute_reply":"2022-10-28T05:17:10.142172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntarget = []\n# train = []\nfor i in range(num):\n    target.append(pd.get_dummies(train[i]['label']))\n#     target.append(train_list[i][['team_A_scoring_within_10sec','team_B_scoring_within_10sec']])\n#     train.append(train_list[i].drop(['game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'team_A_scoring_within_10sec', 'team_B_scoring_within_10sec', 'label'], axis = 1))\n    train[i] = train[i].drop(['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec', 'label'], axis = 1)\n\nfor i in range(num):\n    target[i].columns = ['nobody_scores', 'team_A_scores', 'team_B_scores']\n    \ntest = test.drop(['id'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:10.144693Z","iopub.execute_input":"2022-10-28T05:17:10.144991Z","iopub.status.idle":"2022-10-28T05:17:18.297383Z","shell.execute_reply.started":"2022-10-28T05:17:10.144963Z","shell.execute_reply":"2022-10-28T05:17:18.296165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_list = test.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:18.298955Z","iopub.execute_input":"2022-10-28T05:17:18.299376Z","iopub.status.idle":"2022-10-28T05:17:18.305030Z","shell.execute_reply.started":"2022-10-28T05:17:18.299340Z","shell.execute_reply":"2022-10-28T05:17:18.303852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.feature_selection import mutual_info_regression\n\n# def make_mi_scores(X, y):\n#     mi_scores = mutual_info_regression(X, y)\n#     mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=X.columns)\n#     mi_scores = mi_scores.sort_values(ascending=False)\n#     return mi_scores\n\n# mi_scores = make_mi_scores(train[0].head(500000), target[0]['team_B_scores'].head(500000))\n# mi_scores","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:18.306834Z","iopub.execute_input":"2022-10-28T05:17:18.307158Z","iopub.status.idle":"2022-10-28T05:17:18.315792Z","shell.execute_reply.started":"2022-10-28T05:17:18.307128Z","shell.execute_reply":"2022-10-28T05:17:18.314848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mi_scores.tail(20)\n# p_to_ball, ball_pos, ball_pos_z, p_vel_z, p_pos_x","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:18.317473Z","iopub.execute_input":"2022-10-28T05:17:18.318104Z","iopub.status.idle":"2022-10-28T05:17:18.327892Z","shell.execute_reply.started":"2022-10-28T05:17:18.318071Z","shell.execute_reply":"2022-10-28T05:17:18.326338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scaler = StandardScaler()\nscaler = SKWrapper(StandardScaler(), variables=train[0].columns.tolist())\nscaler.fit(train[0])\nfor i in range(num):\n    train[i] = scaler.transform(train[i])\ntest = scaler.transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:18.329968Z","iopub.execute_input":"2022-10-28T05:17:18.330397Z","iopub.status.idle":"2022-10-28T05:17:34.241689Z","shell.execute_reply.started":"2022-10-28T05:17:18.330361Z","shell.execute_reply":"2022-10-28T05:17:34.240409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_valid, y_train, y_valid = train_test_split(train[0], target[0], test_size = 0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:34.243310Z","iopub.execute_input":"2022-10-28T05:17:34.244139Z","iopub.status.idle":"2022-10-28T05:17:34.248883Z","shell.execute_reply.started":"2022-10-28T05:17:34.244102Z","shell.execute_reply":"2022-10-28T05:17:34.247799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model evaluation**\n\nThanks to @WEI XIE for Catboost details :)","metadata":{}},{"cell_type":"code","source":"import catboost\n\nMAX_ITER = 6000\nPATIENCE = 100\nDISPLAY_FREQ = 100\n\nMODEL_PARAMS = {'random_seed': 1234,    \n                'learning_rate': 0.01,                \n                'iterations': MAX_ITER,\n                'early_stopping_rounds': PATIENCE,\n                'metric_period': DISPLAY_FREQ,\n                'use_best_model': True,\n                'eval_metric': 'Logloss',\n                'task_type': 'GPU'\n               }\n\n\npredsA = []\npredsB = []\n\nfor i in range(num-1):\n    mdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\n    mdl.fit(X=train[i], y=target[i]['team_A_scores'],\n          eval_set=[(train[num-1], target[num-1]['team_A_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\n    predsA.append(mdl.predict_proba(test)[:,1].T)\n    mdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\n    mdl.fit(X=mirror_x(train[i], col_list), y=target[i]['team_A_scores'],\n          eval_set=[(train[num-1], target[num-1]['team_A_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\n    predsA.append(mdl.predict_proba(test)[:,1].T)\n    \n    \n#     mdl.fit(X=change_players(train[i], col_list), y=target[i]['team_A_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n#     predsA.append(mdl.predict_proba(test)[:,1].T)\n#     mdl.fit(X=change_players(change_players(train[i], col_list), col_list), y=target[i]['team_A_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n#     predsA.append(mdl.predict_proba(test)[:,1].T)\n    \n#     mdl.fit(X=make_inv(train[i], col_list), y=target[i]['team_B_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n#     predsA.append(mdl.predict_proba(test)[:,1].T)\n\n\nfor i in range(num-1):\n    mdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\n    mdl.fit(X=train[i], y=target[i]['team_B_scores'],\n          eval_set=[(train[num-1], target[num-1]['team_B_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\n    predsB.append(mdl.predict_proba(test)[:,1].T)\n    mdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\n    mdl.fit(X=mirror_x(train[i], col_list), y=target[i]['team_B_scores'],\n          eval_set=[(train[num-1], target[num-1]['team_B_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\n    predsB.append(mdl.predict_proba(test)[:,1].T)\n    \n    \n#     mdl.fit(X=change_players(train[i],col_list), y=target[i]['team_B_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n#     predsB.append(mdl.predict_proba(test)[:,1].T)\n#     mdl.fit(X=change_players(change_players(train[i],col_list),col_list), y=target[i]['team_B_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n#     predsB.append(mdl.predict_proba(test)[:,1].T)\n    \n#     mdl.fit(X=make_inv(train[i],col_list), y=target[i]['team_A_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n#     predsB.append(mdl.predict_proba(test)[:,1].T)\n    \n#  ++   \n    \nmdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\nmdl.fit(X=train[num-1], y=target[num-1]['team_A_scores'],\n          eval_set=[(train[0], target[0]['team_A_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\npredsA.append(mdl.predict_proba(test)[:,1].T)\nmdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\nmdl.fit(X=mirror_x(train[num-1], col_list), y=target[num-1]['team_A_scores'],\n          eval_set=[(train[0], target[0]['team_A_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\npredsA.append(mdl.predict_proba(test)[:,1].T)\n\n# mdl.fit(X=change_players(train[num-1],col_list), y=target[num-1]['team_A_scores'],\n#           eval_set=[(train[0], target[0]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# predsA.append(mdl.predict_proba(test)[:,1].T)\n# mdl.fit(X=change_players(change_players(train[num-1],col_list),col_list), y=target[num-1]['team_A_scores'],\n#           eval_set=[(train[0], target[0]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# predsA.append(mdl.predict_proba(test)[:,1].T)\n\n# mdl.fit(X=make_inv(train[num-1], col_list), y=target[num-1]['team_B_scores'],\n#           eval_set=[(train[0], target[0]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# predsA.append(mdl.predict_proba(test)[:,1].T)\n\nmdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\nmdl.fit(X=train[num-1], y=target[num-1]['team_B_scores'],\n          eval_set=[(train[0], target[0]['team_B_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\npredsB.append(mdl.predict_proba(test)[:,1].T)\nmdl = catboost.CatBoostClassifier(**MODEL_PARAMS)\nmdl.fit(X=mirror_x(train[num-1], col_list), y=target[num-1]['team_B_scores'],\n          eval_set=[(train[0], target[0]['team_B_scores'])],\n          early_stopping_rounds = PATIENCE,\n          metric_period = DISPLAY_FREQ)\npredsB.append(mdl.predict_proba(test)[:,1].T)\n\n# mdl.fit(X=change_players(train[num-1], col_list), y=target[num-1]['team_B_scores'],\n#           eval_set=[(train[0], target[0]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# predsB.append(mdl.predict_proba(test)[:,1].T)\n# mdl.fit(X=change_players(change_players(train[num-1], col_list),col_list), y=target[num-1]['team_B_scores'],\n#           eval_set=[(train[0], target[0]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# predsB.append(mdl.predict_proba(test)[:,1].T)\n\n# mdl.fit(X=make_inv(train[num-1],col_list), y=target[num-1]['team_A_scores'],\n#           eval_set=[(train[0], target[0]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# predsB.append(mdl.predict_proba(test)[:,1].T)\n\n# --\n\n# mdlA = catboost.CatBoostClassifier(**MODEL_PARAMS)\n# for i in range(num-1):\n#     mdlA.fit(X=train[i], y=target[i]['team_A_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# mdlA.fit(X=train[num-1], y=target[num-1]['team_A_scores'],\n#           eval_set=[(train[0], target[0]['team_A_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n\n# mdlB = catboost.CatBoostClassifier(**MODEL_PARAMS)\n# for i in range(num-1):\n#     mdlB.fit(X=train[i], y=target[i]['team_B_scores'],\n#           eval_set=[(train[num-1], target[num-1]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)\n# mdlB.fit(X=train[num-1], y=target[num-1]['team_B_scores'],\n#           eval_set=[(train[0], target[0]['team_B_scores'])],\n#           early_stopping_rounds = PATIENCE,\n#           metric_period = DISPLAY_FREQ)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:34.250791Z","iopub.execute_input":"2022-10-28T05:17:34.251088Z","iopub.status.idle":"2022-10-28T05:17:34.977851Z","shell.execute_reply.started":"2022-10-28T05:17:34.251060Z","shell.execute_reply":"2022-10-28T05:17:34.976340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from catboost import CatBoost\n\n# model = CatBoost()\n\n# grid = {'learning_rate': [0.5],\n#         'depth': [6],\n#         'l2_leaf_reg': [1, 3, 5, 7]}\n\n# grid_search_result = model.grid_search(grid, \n#                                        X=train[0], \n#                                        y=target[0]['team_A_scores'],\n#                                        cv=3,\n#                                        plot=True, \n#                                        verbose=100)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:34.978932Z","iopub.status.idle":"2022-10-28T05:17:34.980448Z","shell.execute_reply.started":"2022-10-28T05:17:34.980107Z","shell.execute_reply":"2022-10-28T05:17:34.980139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Making prediction**","metadata":{}},{"cell_type":"code","source":"predictionA = np.average(np.array(predsA),axis=0)\npredictionB = np.average(np.array(predsB),axis=0)\n# predictionA = mdlA.predict_proba(test)[:,1].T\n# predictionB = mdlB.predict_proba(test)[:,1].T\n\npreds = pd.DataFrame([predictionA, predictionB])\npreds = preds.T\npreds.columns = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\npreds","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:34.982230Z","iopub.status.idle":"2022-10-28T05:17:34.982786Z","shell.execute_reply.started":"2022-10-28T05:17:34.982503Z","shell.execute_reply":"2022-10-28T05:17:34.982529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample[['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']] = preds[['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']]","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:34.984807Z","iopub.status.idle":"2022-10-28T05:17:34.985391Z","shell.execute_reply.started":"2022-10-28T05:17:34.985087Z","shell.execute_reply":"2022-10-28T05:17:34.985114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T05:17:34.987240Z","iopub.status.idle":"2022-10-28T05:17:34.987785Z","shell.execute_reply.started":"2022-10-28T05:17:34.987501Z","shell.execute_reply":"2022-10-28T05:17:34.987527Z"},"trusted":true},"execution_count":null,"outputs":[]}]}