{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"markdown","source":"## Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np  # linear algebra\nimport pandas as pd  # data manipulation\nimport os  # file navigation\nimport gc  # garbage collection\n\n# visualization\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly import subplots\n\nfrom sklearn.model_selection import cross_validate  # k-fold Cross Validation\nfrom sklearn.preprocessing import LabelEncoder  # output binary encoding\n\nfrom xgboost import XGBClassifier  # Gradient Boosted Tree (XGBoost)\n\nfrom tensorflow.config import list_physical_devices  # check if GPU is available","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-05T14:13:03.240679Z","iopub.execute_input":"2022-10-05T14:13:03.241079Z","iopub.status.idle":"2022-10-05T14:13:10.118883Z","shell.execute_reply.started":"2022-10-05T14:13:03.241045Z","shell.execute_reply":"2022-10-05T14:13:10.117856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"# training and cross validation\nGPU = list_physical_devices('GPU') != []\nN_ESTIMATORS = 1000\nFOLDS = 10\n\n# data loading\nDEBUG = True\nSAMPLE = 0.5\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:10.120943Z","iopub.execute_input":"2022-10-05T14:13:10.121782Z","iopub.status.idle":"2022-10-05T14:13:10.205979Z","shell.execute_reply.started":"2022-10-05T14:13:10.121742Z","shell.execute_reply":"2022-10-05T14:13:10.204824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-05T14:13:10.207549Z","iopub.execute_input":"2022-10-05T14:13:10.208593Z","iopub.status.idle":"2022-10-05T14:13:10.219859Z","shell.execute_reply.started":"2022-10-05T14:13:10.208557Z","shell.execute_reply":"2022-10-05T14:13:10.218905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncol_dtypes = {\n    'game_num': 'int8', 'event_id': 'int8', 'event_time': 'float16',\n    'ball_pos_x': 'float16', 'ball_pos_y': 'float16', 'ball_pos_z': 'float16',\n    'ball_vel_x': 'float16', 'ball_vel_y': 'float16', 'ball_vel_z': 'float16',\n    'p0_pos_x': 'float16', 'p0_pos_y': 'float16', 'p0_pos_z': 'float16',\n    'p0_vel_x': 'float16', 'p0_vel_y': 'float16', 'p0_vel_z': 'float16',\n    'p0_boost': 'float16', 'p1_pos_x': 'float16', 'p1_pos_y': 'float16',\n    'p1_pos_z': 'float16', 'p1_vel_x': 'float16', 'p1_vel_y': 'float16',\n    'p1_vel_z': 'float16', 'p1_boost': 'float16', 'p2_pos_x': 'float16',\n    'p2_pos_y': 'float16', 'p2_pos_z': 'float16', 'p2_vel_x': 'float16',\n    'p2_vel_y': 'float16', 'p2_vel_z': 'float16', 'p2_boost': 'float16',\n    'p3_pos_x': 'float16', 'p3_pos_y': 'float16', 'p3_pos_z': 'float16',\n    'p3_vel_x': 'float16', 'p3_vel_y': 'float16', 'p3_vel_z': 'float16',\n    'p3_boost': 'float16', 'p4_pos_x': 'float16', 'p4_pos_y': 'float16',\n    'p4_pos_z': 'float16', 'p4_vel_x': 'float16', 'p4_vel_y': 'float16',\n    'p4_vel_z': 'float16', 'p4_boost': 'float16', 'p5_pos_x': 'float16',\n    'p5_pos_y': 'float16', 'p5_pos_z': 'float16', 'p5_vel_x': 'float16',\n    'p5_vel_y': 'float16', 'p5_vel_z': 'float16', 'p5_boost': 'float16',\n    'boost0_timer': 'float16', 'boost1_timer': 'float16', 'boost2_timer': 'float16',\n    'boost3_timer': 'float16', 'boost4_timer': 'float16', 'boost5_timer': 'float16',\n    'player_scoring_next': 'O', 'team_scoring_next': 'O', 'team_A_scoring_within_10sec': 'O',\n    'team_B_scoring_within_10sec': 'O'\n}\ncols = list(col_dtypes.keys())","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:10.222799Z","iopub.execute_input":"2022-10-05T14:13:10.223171Z","iopub.status.idle":"2022-10-05T14:13:10.233149Z","shell.execute_reply.started":"2022-10-05T14:13:10.223135Z","shell.execute_reply":"2022-10-05T14:13:10.232099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_to_data = '../input/tabular-playground-series-oct-2022'\ndf = pd.DataFrame({}, columns=cols)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:10.234765Z","iopub.execute_input":"2022-10-05T14:13:10.235589Z","iopub.status.idle":"2022-10-05T14:13:10.250168Z","shell.execute_reply.started":"2022-10-05T14:13:10.235553Z","shell.execute_reply":"2022-10-05T14:13:10.249222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(10):\n    df_tmp = pd.read_csv(f'{path_to_data}/train_{i}.csv', dtype=col_dtypes)\n    if SAMPLE < 1:\n        df_tmp = df_tmp.sample(frac=SAMPLE, random_state=SEED)\n        \n    df = pd.concat([df, df_tmp])\n    del df_tmp\n    gc.collect()\n    if DEBUG:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:10.251662Z","iopub.execute_input":"2022-10-05T14:13:10.252107Z","iopub.status.idle":"2022-10-05T14:13:41.578530Z","shell.execute_reply.started":"2022-10-05T14:13:10.252073Z","shell.execute_reply":"2022-10-05T14:13:41.577493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:41.580330Z","iopub.execute_input":"2022-10-05T14:13:41.581009Z","iopub.status.idle":"2022-10-05T14:13:42.045577Z","shell.execute_reply.started":"2022-10-05T14:13:41.580972Z","shell.execute_reply":"2022-10-05T14:13:42.044438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"position % gravity centre\ndistance % gate1\ndistance % gate2\nacceleration: derivative (dt) of velocity\n...","metadata":{}},{"cell_type":"code","source":"df[(df['game_num']==27) & (df['event_id'] == 121)]","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:32:37.045652Z","iopub.execute_input":"2022-10-05T14:32:37.046603Z","iopub.status.idle":"2022-10-05T14:32:37.211980Z","shell.execute_reply.started":"2022-10-05T14:32:37.046567Z","shell.execute_reply":"2022-10-05T14:32:37.211079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"this competition is so interesting, I wanted to formulate it as a timeseries problem but the test dataset can't really allow that to happen xD","metadata":{}},{"cell_type":"markdown","source":"# 👀 Quick EDA","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:42.047315Z","iopub.execute_input":"2022-10-05T14:13:42.048044Z","iopub.status.idle":"2022-10-05T14:13:42.603887Z","shell.execute_reply.started":"2022-10-05T14:13:42.048001Z","shell.execute_reply":"2022-10-05T14:13:42.602887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_cols = [\n    'ball_pos_x', 'ball_pos_y', 'ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', \n    'p0_pos_x', 'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', \n    'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z',\n    'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y', 'p2_vel_z',\n    'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x', 'p3_vel_y', 'p3_vel_z',\n    'p4_pos_x', 'p4_pos_y', 'p4_pos_z', 'p4_vel_x', 'p4_vel_y', 'p4_vel_z',\n    'p5_pos_x', 'p5_pos_y', 'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z',\n    'p0_boost', 'p1_boost',  'p2_boost', 'p3_boost', 'p4_boost', 'p5_boost',\n    'boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer', 'boost4_timer', 'boost5_timer'\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:42.605572Z","iopub.execute_input":"2022-10-05T14:13:42.605998Z","iopub.status.idle":"2022-10-05T14:13:42.613104Z","shell.execute_reply.started":"2022-10-05T14:13:42.605956Z","shell.execute_reply":"2022-10-05T14:13:42.611959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_cols = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:42.618430Z","iopub.execute_input":"2022-10-05T14:13:42.618733Z","iopub.status.idle":"2022-10-05T14:13:42.623409Z","shell.execute_reply.started":"2022-10-05T14:13:42.618682Z","shell.execute_reply":"2022-10-05T14:13:42.622279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Input Variables","metadata":{}},{"cell_type":"code","source":"def int_to_grid_coord(k, n):\n    return (k // n) + 1, (k % n) + 1","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:42.625248Z","iopub.execute_input":"2022-10-05T14:13:42.626048Z","iopub.status.idle":"2022-10-05T14:13:42.634675Z","shell.execute_reply.started":"2022-10-05T14:13:42.626003Z","shell.execute_reply":"2022-10-05T14:13:42.633098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_distributions(df, row_count, col_count, title, height):\n    features = df.columns\n    fig = subplots.make_subplots(\n        rows=row_count, cols=col_count,\n        subplot_titles=features\n    )\n\n    for k, col in enumerate(features):\n        i, j = int_to_grid_coord(k, col_count)\n\n        fig.add_trace(\n            go.Histogram(\n                x=df[col].astype('float32'),\n                name=col\n            ),\n            row=i, col=j\n        )\n\n    fig.update_layout(\n        title=title,\n        height=row_count * height,\n        showlegend=False\n    )\n\n    return fig","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:42.637190Z","iopub.execute_input":"2022-10-05T14:13:42.637570Z","iopub.status.idle":"2022-10-05T14:13:42.646079Z","shell.execute_reply.started":"2022-10-05T14:13:42.637543Z","shell.execute_reply":"2022-10-05T14:13:42.644830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distributions(df[input_cols].sample(frac=0.0005), 9, 6, \"Input Variables Distributions\", 300)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:42.647415Z","iopub.execute_input":"2022-10-05T14:13:42.648456Z","iopub.status.idle":"2022-10-05T14:13:44.021949Z","shell.execute_reply.started":"2022-10-05T14:13:42.648417Z","shell.execute_reply":"2022-10-05T14:13:44.021013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Output Variables","metadata":{}},{"cell_type":"code","source":"plot_distributions(df[output_cols].sample(frac=0.0005), 1, 2, \"Output Variables Distributions\", height=600)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:44.023079Z","iopub.execute_input":"2022-10-05T14:13:44.023633Z","iopub.status.idle":"2022-10-05T14:13:44.104131Z","shell.execute_reply.started":"2022-10-05T14:13:44.023594Z","shell.execute_reply":"2022-10-05T14:13:44.103111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"Let's derive 2 new features:\n* For each player (and the ball) let's get their velocity's magnitude\n* For each player, let's get their distance from the ball  \n\nFor both of these we'll need to get the 3D euclidian norm of a vector:\n$$ \\| \\overrightarrow{v} \\| = \\sqrt{x^2 + y^2 + z^2} $$\nWe'll use numpy's linalg.norm() method for that.","metadata":{}},{"cell_type":"code","source":"def euclidian_norm(x):\n    return np.linalg.norm(x, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:44.105660Z","iopub.execute_input":"2022-10-05T14:13:44.106313Z","iopub.status.idle":"2022-10-05T14:13:44.111188Z","shell.execute_reply.started":"2022-10-05T14:13:44.106273Z","shell.execute_reply":"2022-10-05T14:13:44.110218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's group the x, y and z variables by player and ball categories\n# this simplifies the code for the euclidian norm calculation\nvel_groups = {\n    f\"{el}_vel\": [f'{el}_vel_x', f'{el}_vel_y', f'{el}_vel_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups = {\n    f\"{el}_pos\": [f'{el}_pos_x', f'{el}_pos_y', f'{el}_pos_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\nvel_groups","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:44.112585Z","iopub.execute_input":"2022-10-05T14:13:44.113196Z","iopub.status.idle":"2022-10-05T14:13:44.124824Z","shell.execute_reply.started":"2022-10-05T14:13:44.113160Z","shell.execute_reply":"2022-10-05T14:13:44.123865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vec in vel_groups.items():\n    df[col] = euclidian_norm(df[vec])","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:44.126959Z","iopub.execute_input":"2022-10-05T14:13:44.127210Z","iopub.status.idle":"2022-10-05T14:13:45.571297Z","shell.execute_reply.started":"2022-10-05T14:13:44.127186Z","shell.execute_reply":"2022-10-05T14:13:45.570248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pos_groups","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:45.572872Z","iopub.execute_input":"2022-10-05T14:13:45.573521Z","iopub.status.idle":"2022-10-05T14:13:45.582039Z","shell.execute_reply.started":"2022-10-05T14:13:45.573482Z","shell.execute_reply":"2022-10-05T14:13:45.580934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vec in pos_groups.items():\n    df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:45.583759Z","iopub.execute_input":"2022-10-05T14:13:45.584502Z","iopub.status.idle":"2022-10-05T14:13:47.365298Z","shell.execute_reply.started":"2022-10-05T14:13:45.584461Z","shell.execute_reply":"2022-10-05T14:13:47.364329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G = np.zeros((df[vec].shape[0], df[vec].shape[1]))\nprint(G.shape)\nfor vec in pos_groups.values():\n    print(df[vec].shape)\n    G = G + df[vec].values\n    \nG = G/7\nG","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:47.382751Z","iopub.execute_input":"2022-10-05T14:13:47.385368Z","iopub.status.idle":"2022-10-05T14:13:47.808593Z","shell.execute_reply.started":"2022-10-05T14:13:47.385329Z","shell.execute_reply":"2022-10-05T14:13:47.807613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"i calculated a centre of gravity of all players + ball, i think it can be a good new feature!!","metadata":{}},{"cell_type":"code","source":"G = pd.DataFrame(G)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:47.813034Z","iopub.execute_input":"2022-10-05T14:13:47.815354Z","iopub.status.idle":"2022-10-05T14:13:47.822296Z","shell.execute_reply.started":"2022-10-05T14:13:47.815317Z","shell.execute_reply":"2022-10-05T14:13:47.821273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vec in pos_groups.items():\n    df[col + \"centre_gravity_dist\"] = euclidian_norm(df[vec].values - G.values)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:47.827321Z","iopub.execute_input":"2022-10-05T14:13:47.828071Z","iopub.status.idle":"2022-10-05T14:13:48.471601Z","shell.execute_reply.started":"2022-10-05T14:13:47.828033Z","shell.execute_reply":"2022-10-05T14:13:48.470578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T14:13:48.473358Z","iopub.execute_input":"2022-10-05T14:13:48.474058Z","iopub.status.idle":"2022-10-05T14:14:04.003550Z","shell.execute_reply.started":"2022-10-05T14:13:48.474019Z","shell.execute_reply":"2022-10-05T14:14:04.002398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧹 Cleaning","metadata":{}},{"cell_type":"markdown","source":"We drop the columns below because they should not influence the results.  \ngame_num, event_id and event_time are irrelevant.  \nplayer_scoring_next and team_scoring next are a form of data leakage, as in they're synonymous with the output variable.  \nball_pos_ball_dist is always 0, the distance between the ball and itself.","metadata":{}},{"cell_type":"markdown","source":"## Dropping columns","metadata":{}},{"cell_type":"code","source":"cols_to_drop = [\n    'game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'ball_pos_ball_dist'\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:26.629600Z","iopub.execute_input":"2022-10-05T11:44:26.630064Z","iopub.status.idle":"2022-10-05T11:44:26.635239Z","shell.execute_reply.started":"2022-10-05T11:44:26.630025Z","shell.execute_reply":"2022-10-05T11:44:26.634259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(columns=cols_to_drop)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:26.636782Z","iopub.execute_input":"2022-10-05T11:44:26.637355Z","iopub.status.idle":"2022-10-05T11:44:26.979368Z","shell.execute_reply.started":"2022-10-05T11:44:26.637319Z","shell.execute_reply":"2022-10-05T11:44:26.978350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:26.980996Z","iopub.execute_input":"2022-10-05T11:44:26.981410Z","iopub.status.idle":"2022-10-05T11:44:27.203598Z","shell.execute_reply.started":"2022-10-05T11:44:26.981372Z","shell.execute_reply":"2022-10-05T11:44:27.202442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Imputing missing values","metadata":{}},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:27.210508Z","iopub.execute_input":"2022-10-05T11:44:27.211178Z","iopub.status.idle":"2022-10-05T11:44:27.574536Z","shell.execute_reply.started":"2022-10-05T11:44:27.211140Z","shell.execute_reply":"2022-10-05T11:44:27.572638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_p0_pos_x_count = df['p0_pos_x'].isna().sum()\nnull_p0_pos_x_perc = null_p0_pos_x_count / df.shape[0]\nprint(f\"Missing {null_p0_pos_x_count} values ({null_p0_pos_x_perc:.2%})\")","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:27.576011Z","iopub.execute_input":"2022-10-05T11:44:27.576388Z","iopub.status.idle":"2022-10-05T11:44:27.587390Z","shell.execute_reply.started":"2022-10-05T11:44:27.576352Z","shell.execute_reply":"2022-10-05T11:44:27.586378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To keep things simple, let's fill missing values with the mean.","metadata":{}},{"cell_type":"code","source":"def fill_na(df):\n    for i in range(6):\n        df['p'+str(i)+'_pos_x'].fillna(value=df['p'+str(i)+'_pos_x'].mean(), inplace=True)\n        df['p'+str(i)+'_pos_y'].fillna(value=df['p'+str(i)+'_pos_y'].mean(), inplace=True)\n        df['p'+str(i)+'_pos_z'].fillna(value=df['p'+str(i)+'_pos_z'].mean(), inplace=True)\n        df['p'+str(i)+'_vel_x'].fillna(value=df['p'+str(i)+'_vel_x'].mean(), inplace=True)\n        df['p'+str(i)+'_vel_y'].fillna(value=df['p'+str(i)+'_vel_y'].mean(), inplace=True)\n        df['p'+str(i)+'_vel_z'].fillna(value=df['p'+str(i)+'_vel_z'].mean(), inplace=True)\n        df['p'+str(i)+'_boost'].fillna(value=df['p'+str(i)+'_boost'].mean(), inplace=True)\n        df['p'+str(i)+'_vel'].fillna(value=df['p'+str(i)+'_vel'].mean(), inplace=True)\n        df['p'+str(i)+'_pos_ball_dist'].fillna(value=df['p'+str(i)+'_pos_ball_dist'].mean(), inplace=True)\n        df['p'+str(i)+'_poscentre_gravity_dist'].fillna(value=df['p'+str(i)+'_poscentre_gravity_dist'].mean(), inplace=True)\n    df['ball_poscentre_gravity_dist'].fillna(value=df['ball_poscentre_gravity_dist'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:27.597120Z","iopub.execute_input":"2022-10-05T11:44:27.598122Z","iopub.status.idle":"2022-10-05T11:44:27.610550Z","shell.execute_reply.started":"2022-10-05T11:44:27.598087Z","shell.execute_reply":"2022-10-05T11:44:27.609498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fill_na(df)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:27.611868Z","iopub.execute_input":"2022-10-05T11:44:27.612343Z","iopub.status.idle":"2022-10-05T11:44:28.349292Z","shell.execute_reply.started":"2022-10-05T11:44:27.612302Z","shell.execute_reply":"2022-10-05T11:44:28.348300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:28.350622Z","iopub.execute_input":"2022-10-05T11:44:28.351526Z","iopub.status.idle":"2022-10-05T11:44:28.587968Z","shell.execute_reply.started":"2022-10-05T11:44:28.351474Z","shell.execute_reply":"2022-10-05T11:44:28.586928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:28.589809Z","iopub.execute_input":"2022-10-05T11:44:28.590179Z","iopub.status.idle":"2022-10-05T11:44:28.754086Z","shell.execute_reply.started":"2022-10-05T11:44:28.590143Z","shell.execute_reply":"2022-10-05T11:44:28.752677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🚀 Model Training","metadata":{}},{"cell_type":"markdown","source":"We'll be predicting the probability of team A scoring and team B scoring with 2 separate models.","metadata":{}},{"cell_type":"markdown","source":"## Model A","metadata":{}},{"cell_type":"code","source":"# used to encode the binary classes\nle_a = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:28.755792Z","iopub.execute_input":"2022-10-05T11:44:28.756177Z","iopub.status.idle":"2022-10-05T11:44:28.761325Z","shell.execute_reply.started":"2022-10-05T11:44:28.756138Z","shell.execute_reply":"2022-10-05T11:44:28.760017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since our classes our imbalanced, we assign assign a heavier weight to the less frequent class\nclass_counts = df['team_A_scoring_within_10sec'].value_counts()\nscale_pos_weight_a = np.sqrt(class_counts['0'] / class_counts['1'])","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:28.762646Z","iopub.execute_input":"2022-10-05T11:44:28.763320Z","iopub.status.idle":"2022-10-05T11:44:28.808322Z","shell.execute_reply.started":"2022-10-05T11:44:28.763283Z","shell.execute_reply":"2022-10-05T11:44:28.807362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_a = XGBClassifier(\n    n_estimators=N_ESTIMATORS,\n    learning_rate=0.05,\n    objective='binary:logistic',\n    tree_method='gpu_hist' if GPU else 'hist'\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:28.809581Z","iopub.execute_input":"2022-10-05T11:44:28.810699Z","iopub.status.idle":"2022-10-05T11:44:28.819926Z","shell.execute_reply.started":"2022-10-05T11:44:28.810660Z","shell.execute_reply":"2022-10-05T11:44:28.818851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_a = cross_validate(\n    model_a, \n    X=df.drop(columns=['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']).values,\n    y=le_a.fit_transform(df['team_A_scoring_within_10sec'].values),\n    scoring=\"neg_log_loss\",\n    cv=FOLDS,\n    verbose=2,\n    return_estimator=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:44:28.821308Z","iopub.execute_input":"2022-10-05T11:44:28.821730Z","iopub.status.idle":"2022-10-05T11:48:57.428016Z","shell.execute_reply.started":"2022-10-05T11:44:28.821695Z","shell.execute_reply":"2022-10-05T11:48:57.427013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model B","metadata":{}},{"cell_type":"code","source":"# used to encode the binary classes\nle_b = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:48:57.429759Z","iopub.execute_input":"2022-10-05T11:48:57.430472Z","iopub.status.idle":"2022-10-05T11:48:57.435578Z","shell.execute_reply.started":"2022-10-05T11:48:57.430430Z","shell.execute_reply":"2022-10-05T11:48:57.434261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since our classes our imbalanced, we assign assign a heavier weight to the less frequent class\nclass_counts = df['team_B_scoring_within_10sec'].value_counts()\nscale_pos_weight_b = np.sqrt(class_counts['0'] / class_counts['1'])","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:48:57.436882Z","iopub.execute_input":"2022-10-05T11:48:57.437400Z","iopub.status.idle":"2022-10-05T11:48:57.485668Z","shell.execute_reply.started":"2022-10-05T11:48:57.437366Z","shell.execute_reply":"2022-10-05T11:48:57.484846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_b = XGBClassifier(\n    n_estimators=N_ESTIMATORS,\n    learning_rate=0.05,\n    objective='binary:logistic',\n    tree_method='gpu_hist' if GPU else 'hist'\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:48:57.486957Z","iopub.execute_input":"2022-10-05T11:48:57.487295Z","iopub.status.idle":"2022-10-05T11:48:57.493686Z","shell.execute_reply.started":"2022-10-05T11:48:57.487261Z","shell.execute_reply":"2022-10-05T11:48:57.491582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_b = cross_validate(\n    model_b, \n    X=df.drop(columns=['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']).values,\n    y=le_b.fit_transform(df['team_B_scoring_within_10sec'].values),\n    scoring=\"neg_log_loss\",\n    cv=FOLDS,\n    verbose=2,\n    return_estimator=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:48:57.495072Z","iopub.execute_input":"2022-10-05T11:48:57.495429Z","iopub.status.idle":"2022-10-05T11:53:29.788034Z","shell.execute_reply.started":"2022-10-05T11:48:57.495396Z","shell.execute_reply":"2022-10-05T11:53:29.787079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ✔️ Model Evaluation","metadata":{}},{"cell_type":"markdown","source":"## Cross Validation Test Score","metadata":{}},{"cell_type":"markdown","source":"Let's visualize the log loss score on each of our folds, for both our models","metadata":{}},{"cell_type":"code","source":"df_cv_a = pd.DataFrame(\n    {\n        \"model\": \"Model A\",\n        \"fold\": list(range(FOLDS)),\n        \"test_log_loss\": - cv_a[\"test_score\"]\n    }\n)\ndf_cv_b = pd.DataFrame(\n    {\n        \"model\": \"Model B\",\n        \"fold\": list(range(FOLDS)),\n        \"test_log_loss\": - cv_b[\"test_score\"]\n    }\n)\ndf_cv = pd.concat([df_cv_a, df_cv_b])\n\ndel df_cv_a\ndel df_cv_b\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:29.789819Z","iopub.execute_input":"2022-10-05T11:53:29.790450Z","iopub.status.idle":"2022-10-05T11:53:30.176196Z","shell.execute_reply.started":"2022-10-05T11:53:29.790414Z","shell.execute_reply":"2022-10-05T11:53:30.175119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.bar(\n    df_cv, x='fold', y='test_log_loss', color='model', \n    barmode='group', title='Cross Validation Log Loss'\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:30.181055Z","iopub.execute_input":"2022-10-05T11:53:30.181832Z","iopub.status.idle":"2022-10-05T11:53:30.265328Z","shell.execute_reply.started":"2022-10-05T11:53:30.181798Z","shell.execute_reply":"2022-10-05T11:53:30.264341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💌 Submission","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\ntest_cols_to_drop = ['ball_pos_ball_dist']\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:30.269585Z","iopub.execute_input":"2022-10-05T11:53:30.271817Z","iopub.status.idle":"2022-10-05T11:53:35.063902Z","shell.execute_reply.started":"2022-10-05T11:53:30.271781Z","shell.execute_reply":"2022-10-05T11:53:35.062689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G = np.zeros((df_test[vec].shape[0], df_test[vec].shape[1]))\nfor vec in pos_groups.values():\n    G = G + df_test[vec].values\nG = pd.DataFrame(G/7)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:35.065546Z","iopub.execute_input":"2022-10-05T11:53:35.065990Z","iopub.status.idle":"2022-10-05T11:53:35.183394Z","shell.execute_reply.started":"2022-10-05T11:53:35.065952Z","shell.execute_reply":"2022-10-05T11:53:35.182202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(df):\n    for col, vec in vel_groups.items():\n        df[col] = euclidian_norm(df[vec])\n    \n    for col, vec in pos_groups.items():\n        df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)\n    \n    for col, vec in pos_groups.items():\n        df[col + \"centre_gravity_dist\"] = euclidian_norm(df[vec].values - G.values)\n    \n    df = df.drop(columns=test_cols_to_drop)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:35.185064Z","iopub.execute_input":"2022-10-05T11:53:35.185460Z","iopub.status.idle":"2022-10-05T11:53:35.192013Z","shell.execute_reply.started":"2022-10-05T11:53:35.185421Z","shell.execute_reply":"2022-10-05T11:53:35.190998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = preprocess(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:35.193698Z","iopub.execute_input":"2022-10-05T11:53:35.194395Z","iopub.status.idle":"2022-10-05T11:53:40.913534Z","shell.execute_reply.started":"2022-10-05T11:53:35.194357Z","shell.execute_reply":"2022-10-05T11:53:40.912497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# take the mean of the predictions made by the k models gotten out of the k-fold cross validation\npred_a = np.zeros(df_test.shape[0])\nfor estimator in cv_a['estimator']:\n    pred_a += estimator.predict_proba(df_test.drop(columns=['id']).values)[:, 1]\n\npred_a /= FOLDS","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:53:40.915111Z","iopub.execute_input":"2022-10-05T11:53:40.915583Z","iopub.status.idle":"2022-10-05T11:56:09.814240Z","shell.execute_reply.started":"2022-10-05T11:53:40.915544Z","shell.execute_reply":"2022-10-05T11:56:09.813451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# take the mean of the predictions made by the k models gotten out of the k-fold cross validation\npred_b = np.zeros(df_test.shape[0])\nfor estimator in cv_b['estimator']:\n    pred_b += estimator.predict_proba(df_test.drop(columns=['id']).values)[:, 1]\n\npred_b /= FOLDS","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:56:09.815489Z","iopub.execute_input":"2022-10-05T11:56:09.818659Z","iopub.status.idle":"2022-10-05T11:58:38.673357Z","shell.execute_reply.started":"2022-10-05T11:56:09.818626Z","shell.execute_reply":"2022-10-05T11:58:38.672485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": df_test['id'],\n        \"team_A_scoring_within_10sec\": pred_a,\n        \"team_B_scoring_within_10sec\": pred_b\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:58:38.676879Z","iopub.execute_input":"2022-10-05T11:58:38.677896Z","iopub.status.idle":"2022-10-05T11:58:38.687024Z","shell.execute_reply.started":"2022-10-05T11:58:38.677853Z","shell.execute_reply":"2022-10-05T11:58:38.685960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:58:38.688672Z","iopub.execute_input":"2022-10-05T11:58:38.689082Z","iopub.status.idle":"2022-10-05T11:58:38.705786Z","shell.execute_reply.started":"2022-10-05T11:58:38.689046Z","shell.execute_reply":"2022-10-05T11:58:38.704542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-05T11:58:38.707489Z","iopub.execute_input":"2022-10-05T11:58:38.707868Z","iopub.status.idle":"2022-10-05T11:58:40.826169Z","shell.execute_reply.started":"2022-10-05T11:58:38.707830Z","shell.execute_reply":"2022-10-05T11:58:40.824841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks for reading so far, Please upvote if this work has been helpful to you 🙏.\n\nThis notebook was copied and edited from https://www.kaggle.com/code/chazzer/rocket-league-xgboost-feat-engineering-cv.","metadata":{}}]}