{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Kaggle TPS Oct 2022 - Rocket League scoring prediction","metadata":{}},{"cell_type":"markdown","source":"## Dataset Description\nThe dataset consists of sequences of snapshots of the state of a Rocket League match, including position and velocity of all players and the ball, as well as extra information. The goal of the competition is to predict -- from a given snapshot in the game -- for each team, the probability that they will score within the next 10 seconds of game time.","metadata":{}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Preprocssing\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler\nfrom sklearn.compose import ColumnTransformer, make_column_transformer\nfrom sklearn.model_selection import train_test_split\n\n# Cloning models\nfrom sklearn.base import clone\n\n# Supervised Learning Methods\n# Logistic Regression\nfrom sklearn.linear_model import LogisticRegressionCV\n\n# XGBoost\nimport xgboost as xgb\n\n# Neural Network\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.metrics import BinaryCrossentropy\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# Hyperparameter Tuning\nfrom scipy.stats import randint, uniform\nfrom sklearn.model_selection import RandomizedSearchCV\n\nimport keras_tuner as kt\n\n# Unsupervised Learning Methods\n# Principal Components Analysis\nfrom sklearn.decomposition import PCA\n\n# Model Evaluation\nfrom sklearn.metrics import confusion_matrix, classification_report, ConfusionMatrixDisplay, roc_auc_score, log_loss\n\n# Plots\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# To get the number of CPUs\nfrom os import cpu_count","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:06.097685Z","iopub.execute_input":"2022-10-30T01:58:06.098250Z","iopub.status.idle":"2022-10-30T01:58:18.742065Z","shell.execute_reply.started":"2022-10-30T01:58:06.098209Z","shell.execute_reply":"2022-10-30T01:58:18.741101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the number of GPUs available for TensorFlow\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:18.743933Z","iopub.execute_input":"2022-10-30T01:58:18.744565Z","iopub.status.idle":"2022-10-30T01:58:19.199162Z","shell.execute_reply.started":"2022-10-30T01:58:18.744528Z","shell.execute_reply":"2022-10-30T01:58:19.197854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = dict()\ny_pred = dict()\ny_hat = dict()\nlog_loss_score = dict()\nteams = ['A', 'B']\nn_cpu = cpu_count()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:19.200976Z","iopub.execute_input":"2022-10-30T01:58:19.201329Z","iopub.status.idle":"2022-10-30T01:58:19.215106Z","shell.execute_reply.started":"2022-10-30T01:58:19.201300Z","shell.execute_reply":"2022-10-30T01:58:19.214128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Num CPUs Available: \", n_cpu)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:19.218336Z","iopub.execute_input":"2022-10-30T01:58:19.218987Z","iopub.status.idle":"2022-10-30T01:58:19.234290Z","shell.execute_reply.started":"2022-10-30T01:58:19.218949Z","shell.execute_reply":"2022-10-30T01:58:19.233074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def avg_log_loss(y_true_a, y_proba_a, y_true_b, y_proba_b):\n    log_loss_a = log_loss(y_true_a, y_proba_a)\n    log_loss_b = log_loss(y_true_b, y_proba_b)\n    avg_log_loss = (log_loss_a + log_loss_b) / 2\n    return avg_log_loss","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:19.236019Z","iopub.execute_input":"2022-10-30T01:58:19.236731Z","iopub.status.idle":"2022-10-30T01:58:19.249935Z","shell.execute_reply.started":"2022-10-30T01:58:19.236694Z","shell.execute_reply":"2022-10-30T01:58:19.248349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import data","metadata":{}},{"cell_type":"code","source":"data_dir = '../input/tabular-playground-series-oct-2022/'","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:19.254594Z","iopub.execute_input":"2022-10-30T01:58:19.255236Z","iopub.status.idle":"2022-10-30T01:58:19.263374Z","shell.execute_reply.started":"2022-10-30T01:58:19.255190Z","shell.execute_reply":"2022-10-30T01:58:19.262346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes_df = pd.read_csv(data_dir + 'train_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\n\ndf_train = list()\nfor i in range(9):\n    df_train.append(pd.read_csv(f'{data_dir}train_{str(i)}.csv', dtype=dtypes))\n\ndf_train = pd.concat(df_train, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:19.265304Z","iopub.execute_input":"2022-10-30T01:58:19.265923Z","iopub.status.idle":"2022-10-30T01:58:49.157064Z","shell.execute_reply.started":"2022-10-30T01:58:19.265885Z","shell.execute_reply":"2022-10-30T01:58:49.156034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Take the 9th training data set as the \"test\" set.","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(f'{data_dir}train_9.csv', dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:58:49.159729Z","iopub.execute_input":"2022-10-30T01:58:49.160363Z","iopub.status.idle":"2022-10-30T01:59:17.263812Z","shell.execute_reply.started":"2022-10-30T01:58:49.160325Z","shell.execute_reply":"2022-10-30T01:59:17.262637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploratory Data Analysis (EDA)","metadata":{}},{"cell_type":"markdown","source":"For `event_id` and `event_time`, they are useful if we model the data in a time series format.","metadata":{}},{"cell_type":"markdown","source":"Drop the Id columns from the dataset.","metadata":{}},{"cell_type":"code","source":"def drop_id_columns(df):\n    # Drop the columns that are not in the Kaggle test set\n    col_to_drop = ['game_num', 'team_scoring_next', 'player_scoring_next']\n    df2 = df.drop(col_to_drop, axis=1)\n    return df2","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:17.265518Z","iopub.execute_input":"2022-10-30T01:59:17.266159Z","iopub.status.idle":"2022-10-30T01:59:17.272236Z","shell.execute_reply.started":"2022-10-30T01:59:17.266118Z","shell.execute_reply":"2022-10-30T01:59:17.271041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = drop_id_columns(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:17.277860Z","iopub.execute_input":"2022-10-30T01:59:17.278762Z","iopub.status.idle":"2022-10-30T01:59:17.776549Z","shell.execute_reply.started":"2022-10-30T01:59:17.278721Z","shell.execute_reply":"2022-10-30T01:59:17.775242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_values_table(df):\n    \"\"\"\n    input: DataFrame\n    return: DataFrame containing info on missing values for each column\n    \"\"\"\n    # Total missing values\n    mis_val = df.isnull().sum()\n\n    # Percentage of missing values\n    mis_val_percent = 100 * df.isnull().sum() / len(df)\n\n    # Make a table with the results\n    mis_val_table = pd.concat([mis_val, mis_val_percent], axis=1)\n\n    # Rename the columns\n    mis_val_table_ren_columns = mis_val_table.rename(\n        columns={0: 'Missing Values', 1: '% of Total Values'})\n\n    # Sort the table by percentage of missing descending\n    mis_val_table_ren_columns = mis_val_table_ren_columns[\n        mis_val_table_ren_columns.iloc[:, 1] != 0].sort_values(\n        '% of Total Values', ascending=False).round(1)\n\n    # Print some summary information\n    print(\"Your selected dataframe has \" + str(df.shape[1]) + \" columns.\\n\"\n          \"There are \" + str(mis_val_table_ren_columns.shape[0]) +\n          \" columns that have missing values.\")\n\n    # Return the dataframe with missing information\n    return mis_val_table_ren_columns","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:17.778514Z","iopub.execute_input":"2022-10-30T01:59:17.785204Z","iopub.status.idle":"2022-10-30T01:59:17.794724Z","shell.execute_reply.started":"2022-10-30T01:59:17.785155Z","shell.execute_reply":"2022-10-30T01:59:17.793663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For simplicity, we drop the records with missing values.","metadata":{}},{"cell_type":"code","source":"print(f'There are {df_train.shape[0]} rows in the training dataset.')\ndf_train = df_train.dropna()\nprint(f'After dropping missing values, there are {df_train.shape[0]} rows in the training dataset.')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:17.795949Z","iopub.execute_input":"2022-10-30T01:59:17.796525Z","iopub.status.idle":"2022-10-30T01:59:18.754430Z","shell.execute_reply.started":"2022-10-30T01:59:17.796489Z","shell.execute_reply":"2022-10-30T01:59:18.753183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scale down the dataset\nThe original dataset takes a snapshot of the game instance at 10 frames per second. We take 1 frame every `n_frames` frames to scale down the dataset.","metadata":{}},{"cell_type":"code","source":"n_frames = 10\ndf_train = df_train.iloc[::n_frames]","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:18.756078Z","iopub.execute_input":"2022-10-30T01:59:18.756481Z","iopub.status.idle":"2022-10-30T01:59:18.763377Z","shell.execute_reply.started":"2022-10-30T01:59:18.756443Z","shell.execute_reply":"2022-10-30T01:59:18.762317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Univariate analysis","metadata":{}},{"cell_type":"markdown","source":"#### Target variables","metadata":{}},{"cell_type":"markdown","source":"We can observe that the dataset is imbalanced. Only about 6% of training records have `team_A_scoring_within_10sec` or `team_B_scoring_within_10sec` = 1.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, figsize=(12,4))\nsns.countplot(data=df_train, y='team_A_scoring_within_10sec', ax=axs[0])\nsns.countplot(data=df_train, y='team_B_scoring_within_10sec', ax=axs[1])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:03:26.018287Z","iopub.execute_input":"2022-10-29T16:03:26.018716Z","iopub.status.idle":"2022-10-29T16:03:26.735485Z","shell.execute_reply.started":"2022-10-29T16:03:26.018678Z","shell.execute_reply":"2022-10-29T16:03:26.734470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ball position and velocities","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, figsize=(8,8))\nfor i, var in enumerate([f'ball_pos_{ax}' for ax in ['x', 'y', 'z']]):\n    sns.histplot(data=df_train[var], ax=axs[i])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:03:26.738784Z","iopub.execute_input":"2022-10-29T16:03:26.739085Z","iopub.status.idle":"2022-10-29T16:03:31.424361Z","shell.execute_reply.started":"2022-10-29T16:03:26.739058Z","shell.execute_reply":"2022-10-29T16:03:31.423404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, figsize=(8,8))\nfor i, var in enumerate([f'ball_vel_{ax}' for ax in ['x', 'y', 'z']]):\n    sns.histplot(data=df_train[var], ax=axs[i])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:03:31.425905Z","iopub.execute_input":"2022-10-29T16:03:31.426955Z","iopub.status.idle":"2022-10-29T16:03:37.536150Z","shell.execute_reply.started":"2022-10-29T16:03:31.426916Z","shell.execute_reply":"2022-10-29T16:03:37.535071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Player position and velocity","metadata":{}},{"cell_type":"markdown","source":"We just visualize the first player for each team i.e. players 0 and 3.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, figsize=(8,8))\nfor i, var in enumerate([f'p0_pos_{ax}' for ax in ['x', 'y', 'z']]):\n    sns.histplot(data=df_train[var], ax=axs[i])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:03:37.537494Z","iopub.execute_input":"2022-10-29T16:03:37.538121Z","iopub.status.idle":"2022-10-29T16:03:45.119439Z","shell.execute_reply.started":"2022-10-29T16:03:37.538076Z","shell.execute_reply":"2022-10-29T16:03:45.118213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, figsize=(8,8))\nfor i, var in enumerate([f'p3_pos_{ax}' for ax in ['x', 'y', 'z']]):\n    sns.histplot(data=df_train[var], ax=axs[i])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:03:45.121066Z","iopub.execute_input":"2022-10-29T16:03:45.121559Z","iopub.status.idle":"2022-10-29T16:03:51.586916Z","shell.execute_reply.started":"2022-10-29T16:03:45.121517Z","shell.execute_reply":"2022-10-29T16:03:51.585852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, figsize=(8,8))\nfor i, var in enumerate([f'p0_vel_{ax}' for ax in ['x', 'y', 'z']]):\n    sns.histplot(data=df_train[var], ax=axs[i])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:03:51.588528Z","iopub.execute_input":"2022-10-29T16:03:51.588916Z","iopub.status.idle":"2022-10-29T16:04:31.252979Z","shell.execute_reply.started":"2022-10-29T16:03:51.588878Z","shell.execute_reply":"2022-10-29T16:04:31.251818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=3, figsize=(8,8))\nfor i, var in enumerate([f'p3_vel_{ax}' for ax in ['x', 'y', 'z']]):\n    sns.histplot(data=df_train[var], ax=axs[i])\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:04:31.254633Z","iopub.execute_input":"2022-10-29T16:04:31.255031Z","iopub.status.idle":"2022-10-29T16:05:11.382190Z","shell.execute_reply.started":"2022-10-29T16:04:31.254991Z","shell.execute_reply":"2022-10-29T16:05:11.381098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Boost timers","metadata":{}},{"cell_type":"markdown","source":"For the boost timer feature, most of the records have value 0 since the orb is already spawned on the field.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1)\nsns.histplot(data=df_train['boost0_timer'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:05:11.383840Z","iopub.execute_input":"2022-10-29T16:05:11.384237Z","iopub.status.idle":"2022-10-29T16:05:12.931596Z","shell.execute_reply.started":"2022-10-29T16:05:11.384198Z","shell.execute_reply":"2022-10-29T16:05:12.930480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Relationship with target variable","metadata":{}},{"cell_type":"markdown","source":"#### Boost timers\n- `boost0_timer` and `boost1_timer` have an effect on team A scoring.\n- `boost2_timer` and `boost3_timer` have no effect.\n- `boost4_timer` and `boost5_timer` have an effect on team B scoring.","metadata":{}},{"cell_type":"code","source":"fig, axs= plt.subplots(ncols=2, nrows=6, figsize=(10, 10))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'boost{i}_timer' for i in range(6)]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(6):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:05:12.933302Z","iopub.execute_input":"2022-10-29T16:05:12.934008Z","iopub.status.idle":"2022-10-29T16:05:18.992897Z","shell.execute_reply.started":"2022-10-29T16:05:12.933965Z","shell.execute_reply":"2022-10-29T16:05:18.991783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ball position\n- When `ball_pos_y` is high, it is more likely that team A will score.\n- When `ball_pos_y` is low, it is more likely that team B will score.\n- The other two axis have no effect.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=3, figsize=(10,4))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'ball_pos_{ax}' for ax in ['x', 'y', 'z']]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(3):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:05:18.994506Z","iopub.execute_input":"2022-10-29T16:05:18.994902Z","iopub.status.idle":"2022-10-29T16:05:20.914742Z","shell.execute_reply.started":"2022-10-29T16:05:18.994865Z","shell.execute_reply":"2022-10-29T16:05:20.913550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plot below shows the density plot of the ball position in the last 5 seconds of each round. From this, we can see that team A's net is centered at `x=0, y=-100` while team B's goal is centered at `x=0, y=100`.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, figsize=(10, 5))\nfor i, team in enumerate(teams):\n    sns.kdeplot(data=df_train[df_train['event_time'] > -5], x='ball_pos_x',\n                y='ball_pos_y', hue='team_' + team + '_scoring_within_10sec', ax=axs[i])\nplt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:05:20.916614Z","iopub.execute_input":"2022-10-29T16:05:20.917398Z","iopub.status.idle":"2022-10-29T16:07:20.464423Z","shell.execute_reply.started":"2022-10-29T16:05:20.917354Z","shell.execute_reply":"2022-10-29T16:07:20.463389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ball velocity\n- When `ball_vel_y` is high, it is more likely that team A will score.\n- When `ball_vel_y` is low, it is more likely that team B will score.\n- The other two axis have no effect.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=3, figsize=(10,4))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'ball_vel_{ax}' for ax in ['x', 'y', 'z']]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(3):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:07:20.465698Z","iopub.execute_input":"2022-10-29T16:07:20.466052Z","iopub.status.idle":"2022-10-29T16:07:22.580185Z","shell.execute_reply.started":"2022-10-29T16:07:20.466014Z","shell.execute_reply":"2022-10-29T16:07:22.579086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Players' position\n- A higher y-coordinate of the player's position is associated with higher probability of team A scoring.\n- A lower y-coordinate of the player's position is associated with higher probability of team B scoring.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=6, figsize=(10,6))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'p{i}_pos_x' for i in range(6)]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(6):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:07:22.581802Z","iopub.execute_input":"2022-10-29T16:07:22.582176Z","iopub.status.idle":"2022-10-29T16:07:26.183602Z","shell.execute_reply.started":"2022-10-29T16:07:22.582139Z","shell.execute_reply":"2022-10-29T16:07:26.182458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=6, figsize=(10,6))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'p{i}_pos_y' for i in range(6)]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(6):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:07:26.184990Z","iopub.execute_input":"2022-10-29T16:07:26.186205Z","iopub.status.idle":"2022-10-29T16:07:29.286743Z","shell.execute_reply.started":"2022-10-29T16:07:26.186166Z","shell.execute_reply":"2022-10-29T16:07:29.285561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=6, figsize=(10,6))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'p{i}_pos_z' for i in range(6)]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(6):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:07:29.294956Z","iopub.execute_input":"2022-10-29T16:07:29.295554Z","iopub.status.idle":"2022-10-29T16:07:44.624711Z","shell.execute_reply.started":"2022-10-29T16:07:29.295494Z","shell.execute_reply":"2022-10-29T16:07:44.623464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Players' boosts\n- When players 3-5 have lower boost remaining, it is more likely that team A will score.\n- When players 0-2 have lower boost remaining, it is more likely that team B will score.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=6, figsize=(10,6))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'p{i}_boost' for i in range(6)]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(6):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:07:44.626446Z","iopub.execute_input":"2022-10-29T16:07:44.626864Z","iopub.status.idle":"2022-10-29T16:07:50.797553Z","shell.execute_reply.started":"2022-10-29T16:07:44.626826Z","shell.execute_reply":"2022-10-29T16:07:50.796527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"### Player's distance to ball\n- Calculate the distance of each player to the ball. \n- Generally when the distance is lower, it is more likely for either team to score, but the difference is small.","metadata":{}},{"cell_type":"code","source":"def add_feature_dist_to_ball(df_):\n    df = df_.copy()\n    ball_pos = df[['ball_pos_x', 'ball_pos_y', 'ball_pos_z']].to_numpy()\n\n    for i in range(6):\n        player_pos = df[[f'p{i}_pos_x',\n                         f'p{i}_pos_y', f'p{i}_pos_z']].to_numpy()\n        vector_to_ball = ball_pos - player_pos\n        dist_to_ball = np.sqrt((vector_to_ball * vector_to_ball).sum(axis=1))\n        df[f'p{i}_dist_to_ball'] = dist_to_ball\n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:18.765519Z","iopub.execute_input":"2022-10-30T01:59:18.765975Z","iopub.status.idle":"2022-10-30T01:59:18.775963Z","shell.execute_reply.started":"2022-10-30T01:59:18.765939Z","shell.execute_reply":"2022-10-30T01:59:18.774858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = add_feature_dist_to_ball(df_train)\ndf_train.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:18.778593Z","iopub.execute_input":"2022-10-30T01:59:18.778951Z","iopub.status.idle":"2022-10-30T01:59:18.948474Z","shell.execute_reply.started":"2022-10-30T01:59:18.778924Z","shell.execute_reply":"2022-10-30T01:59:18.947597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, nrows=6, clear=True, figsize=(10,6))\nfor j, team in enumerate(teams):\n    for i, var in enumerate([f'p{i}_dist_to_ball' for i in range(6)]):\n        sns.boxplot(data=df_train, x=var, y='team_' + team + '_scoring_within_10sec', ax=axs[i, j], orient='h')\n    for i in range(6):\n        axs[i, j].set_ylabel('')\n    axs[0, j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:18.949746Z","iopub.execute_input":"2022-10-30T01:59:18.950639Z","iopub.status.idle":"2022-10-30T01:59:20.693218Z","shell.execute_reply.started":"2022-10-30T01:59:18.950603Z","shell.execute_reply":"2022-10-30T01:59:20.692120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Index of closest player\nWith the distance of each player to the ball, we can calculate the index of the player that's closest to the ball. We can view that player as a 'runner' that has possession of the ball. It should be more likely for the team to have possession to score.","metadata":{}},{"cell_type":"code","source":"def add_feature_closest_player_to_ball(df_):\n    df = df_.copy()\n\n    # If the dist to ball does not exist yet, add it to df\n    added = False\n    if any([(f'p{i}_dist_to_ball' not in list(df.columns)) for i in range(6)]):\n        df = add_feature_dist_to_ball(df)\n        added = True\n\n    player_dists = df[[f'p{i}_dist_to_ball' for i in range(6)]].to_numpy()\n    df['closest_player_to_ball'] = np.argmin(player_dists, axis=1)\n\n    # Drop the dist columns if they are added temporarily\n    if added:\n        df = df.drop([f'p{i}_dist_to_ball' for i in range(6)], axis=1)\n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:20.694774Z","iopub.execute_input":"2022-10-30T01:59:20.695386Z","iopub.status.idle":"2022-10-30T01:59:20.702826Z","shell.execute_reply.started":"2022-10-30T01:59:20.695348Z","shell.execute_reply":"2022-10-30T01:59:20.701803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = add_feature_closest_player_to_ball(df_train)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:20.704440Z","iopub.execute_input":"2022-10-30T01:59:20.704813Z","iopub.status.idle":"2022-10-30T01:59:20.770037Z","shell.execute_reply.started":"2022-10-30T01:59:20.704778Z","shell.execute_reply":"2022-10-30T01:59:20.768916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, figsize=(10,4))\nfor i, team in enumerate(teams):\n    sns.countplot(data=df_train[df_train['team_' + team + '_scoring_within_10sec']==1], x='closest_player_to_ball', hue='team_' + team + '_scoring_within_10sec', ax=axs[i], orient='v')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:20.771887Z","iopub.execute_input":"2022-10-30T01:59:20.772310Z","iopub.status.idle":"2022-10-30T01:59:21.190488Z","shell.execute_reply.started":"2022-10-30T01:59:20.772273Z","shell.execute_reply":"2022-10-30T01:59:21.189545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plot above confirms that if the players 0-2 are closest to the ball, it is more likely that team A will score, vice versa.","metadata":{}},{"cell_type":"markdown","source":"### Ball distance to net","metadata":{}},{"cell_type":"markdown","source":"From EDA, we know that team A's net is centered at `x=0, y=-100`, and team B's net is centered at `x=0, y=100`","metadata":{}},{"cell_type":"code","source":"def add_feature_ball_dist_to_net(df_):\n    df = df_.copy()\n    ball_pos = df[['ball_pos_x', 'ball_pos_y', 'ball_pos_z']].to_numpy()\n\n    for team in teams:\n        if team == 'A':\n            team_net_pos = np.array((0, -100, 0))\n        else:\n            team_net_pos = np.array((0, 100, 0))\n\n        vector_to_net = ball_pos - team_net_pos\n        dist_to_net = np.sqrt((vector_to_net * vector_to_net).sum(axis=1))\n        df[f'ball_dist_to_net_{team}'] = dist_to_net\n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.191846Z","iopub.execute_input":"2022-10-30T01:59:21.192848Z","iopub.status.idle":"2022-10-30T01:59:21.200118Z","shell.execute_reply.started":"2022-10-30T01:59:21.192799Z","shell.execute_reply":"2022-10-30T01:59:21.199207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = add_feature_ball_dist_to_net(df_train)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.201861Z","iopub.execute_input":"2022-10-30T01:59:21.202222Z","iopub.status.idle":"2022-10-30T01:59:21.251207Z","shell.execute_reply.started":"2022-10-30T01:59:21.202187Z","shell.execute_reply":"2022-10-30T01:59:21.250285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, figsize=(10,6))\nfor j, team in enumerate(teams):\n    sns.boxplot(data=df_train, x='ball_dist_to_net_' + team, y='team_' + team + '_scoring_within_10sec', ax=axs[j], orient='h')\n    axs[j].set_title('team_' + team + '_scoring_within_10sec')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.252756Z","iopub.execute_input":"2022-10-30T01:59:21.253120Z","iopub.status.idle":"2022-10-30T01:59:21.638704Z","shell.execute_reply.started":"2022-10-30T01:59:21.253084Z","shell.execute_reply":"2022-10-30T01:59:21.637586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"### Split into X and y","metadata":{}},{"cell_type":"code","source":"target_vars = [f'team_{team}_scoring_within_10sec' for team in teams]\n\nX_train = df_train.drop(target_vars, axis=1)\ny_train = df_train[target_vars]","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.640404Z","iopub.execute_input":"2022-10-30T01:59:21.640799Z","iopub.status.idle":"2022-10-30T01:59:21.663215Z","shell.execute_reply.started":"2022-10-30T01:59:21.640763Z","shell.execute_reply":"2022-10-30T01:59:21.662363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape, y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.664515Z","iopub.execute_input":"2022-10-30T01:59:21.665445Z","iopub.status.idle":"2022-10-30T01:59:21.672293Z","shell.execute_reply.started":"2022-10-30T01:59:21.665410Z","shell.execute_reply":"2022-10-30T01:59:21.671307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data normalization\nWe normalize the numerical variables:\n- We drop the `ball_pos_x/y/z` features and use the ball's distance to team A's net instead.\n- We drop the players' position and use the players' distance to the ball instead.\n- We also drop `boost2_timer` and `boost3_timer` as they are not likely to be useful according to EDA.\n- For the remaining features, we use the `MinMaxScaler` to preserve the shape of the data.","metadata":{}},{"cell_type":"code","source":"drop_cols = ['event_time', 'event_id'] + [f'p{i}_pos_{ax}' for i in range(6) for ax in ['x', 'y', 'z']] + [f'p{i}_vel_{ax}' for i in range(\n    6) for ax in ['x', 'z']] + [f'ball_pos_{ax}' for ax in ['x', 'y', 'z']] + ['ball_dist_to_net_B'] + ['ball_vel_x', 'ball_vel_z', 'boost2_timer', 'boost3_timer']\n\nremain_cols = [x for x in X_train.columns if x not in drop_cols]\n\nct = make_column_transformer(\n    ('drop', drop_cols),\n    (MinMaxScaler(feature_range=(-1, 1)), remain_cols),\n    verbose_feature_names_out=False\n)\n\nct.fit(X_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.673745Z","iopub.execute_input":"2022-10-30T01:59:21.674249Z","iopub.status.idle":"2022-10-30T01:59:21.764691Z","shell.execute_reply.started":"2022-10-30T01:59:21.674215Z","shell.execute_reply":"2022-10-30T01:59:21.763608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform feature scaling on X_train\nX_train = ct.transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.765961Z","iopub.execute_input":"2022-10-30T01:59:21.766331Z","iopub.status.idle":"2022-10-30T01:59:21.819169Z","shell.execute_reply.started":"2022-10-30T01:59:21.766296Z","shell.execute_reply":"2022-10-30T01:59:21.818143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transform the test dataset","metadata":{}},{"cell_type":"code","source":"# Scale down the dataset\ndf_test = df_test.iloc[::n_frames]","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.820967Z","iopub.execute_input":"2022-10-30T01:59:21.821376Z","iopub.status.idle":"2022-10-30T01:59:21.827625Z","shell.execute_reply.started":"2022-10-30T01:59:21.821336Z","shell.execute_reply":"2022-10-30T01:59:21.826614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop columns and missing values\ndf_test = drop_id_columns(df_test)\ndf_test = df_test.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.833666Z","iopub.execute_input":"2022-10-30T01:59:21.833955Z","iopub.status.idle":"2022-10-30T01:59:21.955554Z","shell.execute_reply.started":"2022-10-30T01:59:21.833929Z","shell.execute_reply":"2022-10-30T01:59:21.954560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature engineering\n# Distance to ball\ndf_test = add_feature_dist_to_ball(df_test)\n\n# Closest player to ball\ndf_test = add_feature_closest_player_to_ball(df_test)\n\n# Ball distance to net\ndf_test = add_feature_ball_dist_to_net(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:21.956951Z","iopub.execute_input":"2022-10-30T01:59:21.957958Z","iopub.status.idle":"2022-10-30T01:59:22.120038Z","shell.execute_reply.started":"2022-10-30T01:59:21.957921Z","shell.execute_reply":"2022-10-30T01:59:22.119023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create X and y\nX_test = df_test.drop(target_vars, axis=1)\ny_test = df_test[target_vars]","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:22.121579Z","iopub.execute_input":"2022-10-30T01:59:22.122244Z","iopub.status.idle":"2022-10-30T01:59:22.143820Z","shell.execute_reply.started":"2022-10-30T01:59:22.122205Z","shell.execute_reply":"2022-10-30T01:59:22.142961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature dropping and scaling\nX_test = ct.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:22.145284Z","iopub.execute_input":"2022-10-30T01:59:22.145686Z","iopub.status.idle":"2022-10-30T01:59:22.197184Z","shell.execute_reply.started":"2022-10-30T01:59:22.145635Z","shell.execute_reply":"2022-10-30T01:59:22.196163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Shape of training data:')\nprint(X_train.shape, y_train.shape)\nprint('Shape of test data:')\nprint(X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:22.198573Z","iopub.execute_input":"2022-10-30T01:59:22.198962Z","iopub.status.idle":"2022-10-30T01:59:22.206080Z","shell.execute_reply.started":"2022-10-30T01:59:22.198927Z","shell.execute_reply":"2022-10-30T01:59:22.204720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ct.get_feature_names_out()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:22.207829Z","iopub.execute_input":"2022-10-30T01:59:22.208548Z","iopub.status.idle":"2022-10-30T01:59:22.218393Z","shell.execute_reply.started":"2022-10-30T01:59:22.208513Z","shell.execute_reply":"2022-10-30T01:59:22.217326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression Model","metadata":{}},{"cell_type":"markdown","source":"We use train a basic logistic regression model with L2-regularizer and built-in cross validation as the baseline.","metadata":{}},{"cell_type":"code","source":"models['logistic_regr'] = dict()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:08:01.431937Z","iopub.execute_input":"2022-10-29T16:08:01.432385Z","iopub.status.idle":"2022-10-29T16:08:01.441141Z","shell.execute_reply.started":"2022-10-29T16:08:01.432346Z","shell.execute_reply":"2022-10-29T16:08:01.440103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfor team in teams:\n    models['logistic_regr'][team] = LogisticRegressionCV(cv=5, random_state=0)\n    models['logistic_regr'][team].fit(X_train, y_train['team_' + team + '_scoring_within_10sec'])","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:08:01.442550Z","iopub.execute_input":"2022-10-29T16:08:01.442849Z","iopub.status.idle":"2022-10-29T16:11:50.173465Z","shell.execute_reply.started":"2022-10-29T16:08:01.442824Z","shell.execute_reply":"2022-10-29T16:11:50.169314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction results","metadata":{}},{"cell_type":"code","source":"y_pred['logistic_regr'] = dict()\ny_hat['logistic_regr'] = dict()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:11:50.175285Z","iopub.execute_input":"2022-10-29T16:11:50.175712Z","iopub.status.idle":"2022-10-29T16:11:50.184601Z","shell.execute_reply.started":"2022-10-29T16:11:50.175671Z","shell.execute_reply":"2022-10-29T16:11:50.182770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for team in teams:\n    y_pred['logistic_regr'][team] = models['logistic_regr'][team].predict(X_test)\n    y_hat['logistic_regr'][team] = models['logistic_regr'][team].predict_proba(X_test)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:11:50.186435Z","iopub.execute_input":"2022-10-29T16:11:50.187086Z","iopub.status.idle":"2022-10-29T16:11:50.298324Z","shell.execute_reply.started":"2022-10-29T16:11:50.187045Z","shell.execute_reply":"2022-10-29T16:11:50.296613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluate the model using log loss and area under the ROC curve.","metadata":{}},{"cell_type":"code","source":"log_loss_score['logistic_regr'] = avg_log_loss(y_test['team_A_scoring_within_10sec'], y_hat['logistic_regr']['A'][:, 1],\n                                               y_test['team_B_scoring_within_10sec'], y_hat['logistic_regr']['B'][:, 1])\nprint('Log loss score for the model: {:.4f}'.format(log_loss_score['logistic_regr']))","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:11:50.300218Z","iopub.execute_input":"2022-10-29T16:11:50.301064Z","iopub.status.idle":"2022-10-29T16:11:50.493195Z","shell.execute_reply.started":"2022-10-29T16:11:50.301014Z","shell.execute_reply":"2022-10-29T16:11:50.492043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for team in teams:\n    print(f\"ROC AUC Score: {roc_auc_score(y_test['team_' + team + '_scoring_within_10sec'], y_hat['logistic_regr'][team][:,1])}\")\n    print(confusion_matrix(y_test['team_' + team + '_scoring_within_10sec'], y_pred['logistic_regr'][team]))","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:11:50.496251Z","iopub.execute_input":"2022-10-29T16:11:50.496557Z","iopub.status.idle":"2022-10-29T16:11:50.808683Z","shell.execute_reply.started":"2022-10-29T16:11:50.496527Z","shell.execute_reply":"2022-10-29T16:11:50.807566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost Model","metadata":{}},{"cell_type":"markdown","source":"### Hyperparameter tuning","metadata":{}},{"cell_type":"code","source":"# Reports the best scores from the searched models\n# https://www.kaggle.com/code/stuarthallows/using-xgboost-with-scikit-learn/notebook\ndef report_best_scores(results, n_top=3):\n    for i in range(1, n_top + 1):\n        candidates = np.flatnonzero(results['rank_test_score'] == i)\n        for candidate in candidates:\n            print(\"Model with rank: {0}\".format(i))\n            print(\"Mean validation score: {0:.4f} (std: {1:.4f})\".format(\n                  results['mean_test_score'][candidate],\n                  results['std_test_score'][candidate]))\n            print(\"Parameters: {0}\".format(results['params'][candidate]))\n            print(\"\")","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:22.220097Z","iopub.execute_input":"2022-10-30T01:59:22.220937Z","iopub.status.idle":"2022-10-30T01:59:22.229879Z","shell.execute_reply.started":"2022-10-30T01:59:22.220897Z","shell.execute_reply":"2022-10-30T01:59:22.228703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Highly recommend using GPU (tree_method='gpu_hist') to speed up the training,\nxgb_model = xgb.XGBClassifier(tree_method='gpu_hist')\n\nparams = {\n    \"colsample_bytree\": uniform(0.7, 0.3),\n    \"gamma\": uniform(0, 0.5),\n    \"learning_rate\": uniform(0.01, 0.5),  # default 0.1\n    \"max_depth\": randint(2, 6),  # default 3\n    \"n_estimators\": randint(100, 200),  # default 100\n    \"subsample\": uniform(0.6, 0.4)\n}\n\nsearch = RandomizedSearchCV(xgb_model, param_distributions=params, random_state=0,\n                            n_iter=50, cv=3, verbose=1, scoring='neg_log_loss')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:22.232093Z","iopub.execute_input":"2022-10-30T01:59:22.233033Z","iopub.status.idle":"2022-10-30T01:59:22.251809Z","shell.execute_reply.started":"2022-10-30T01:59:22.232998Z","shell.execute_reply":"2022-10-30T01:59:22.250676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsearch.fit(X_train, y_train['team_A_scoring_within_10sec'])\nreport_best_scores(search.cv_results_, 1)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T01:59:46.212910Z","iopub.execute_input":"2022-10-30T01:59:46.213347Z","iopub.status.idle":"2022-10-30T02:01:11.644277Z","shell.execute_reply.started":"2022-10-30T01:59:46.213305Z","shell.execute_reply":"2022-10-30T02:01:11.643453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below plots the mean CV scores for each candidate model:","metadata":{}},{"cell_type":"code","source":"cv_scores = pd.DataFrame(search.cv_results_).sort_values(\n    by='rank_test_score')[['rank_test_score', 'mean_test_score', 'std_test_score']]\n\nplt.subplots(figsize=(8,4))\nsns.scatterplot(data=cv_scores, x='rank_test_score',\n             y='mean_test_score')\nplt.title('CV score for each model (higher is better)')\nplt.xlabel('Model Rank')\nplt.ylabel('Negative Log loss')\nplt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:01:11.648036Z","iopub.execute_input":"2022-10-30T02:01:11.649776Z","iopub.status.idle":"2022-10-30T02:01:11.953350Z","shell.execute_reply.started":"2022-10-30T02:01:11.649743Z","shell.execute_reply":"2022-10-30T02:01:11.952422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"%%time\nmodels['xgboost'] = dict()\n\nmodels['xgboost']['A'] = clone(search.best_estimator_)\nmodels['xgboost']['A'].fit(X_train, y_train['team_A_scoring_within_10sec'])\n\n\nmodels['xgboost']['B'] = clone(models['xgboost']['A'])\nmodels['xgboost']['B'].fit(X_train, y_train['team_B_scoring_within_10sec'])\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:01:11.954714Z","iopub.execute_input":"2022-10-30T02:01:11.955119Z","iopub.status.idle":"2022-10-30T02:01:13.012769Z","shell.execute_reply.started":"2022-10-30T02:01:11.955080Z","shell.execute_reply":"2022-10-30T02:01:13.011705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred['xgboost'] = dict()\ny_hat['xgboost'] = dict()\nfor team in teams:\n    y_pred['xgboost'][team] = models['xgboost'][team].predict(X_test)\n    y_hat['xgboost'][team] = models['xgboost'][team].predict_proba(X_test)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:01:13.015274Z","iopub.execute_input":"2022-10-30T02:01:13.015769Z","iopub.status.idle":"2022-10-30T02:01:14.533514Z","shell.execute_reply.started":"2022-10-30T02:01:13.015731Z","shell.execute_reply":"2022-10-30T02:01:14.532703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_loss_score['xgboost'] = avg_log_loss(y_test['team_A_scoring_within_10sec'], y_hat['xgboost']['A'][:, 1],\n                                                        y_test['team_B_scoring_within_10sec'], y_hat['xgboost']['B'][:, 1])\nprint('Log loss score for the model: {:.4f}'.format(\n    log_loss_score['xgboost']))\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:01:14.534745Z","iopub.execute_input":"2022-10-30T02:01:14.535366Z","iopub.status.idle":"2022-10-30T02:01:14.598552Z","shell.execute_reply.started":"2022-10-30T02:01:14.535329Z","shell.execute_reply":"2022-10-30T02:01:14.597513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for team in teams:\n    print(f\"ROC AUC Score: {roc_auc_score(y_test['team_' + team + '_scoring_within_10sec'], y_hat['xgboost'][team][:,1])}\")\n    print(classification_report(y_test['team_' + team + '_scoring_within_10sec'], y_pred['xgboost'][team]))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:01:14.600048Z","iopub.execute_input":"2022-10-30T02:01:14.600422Z","iopub.status.idle":"2022-10-30T02:01:15.070315Z","shell.execute_reply.started":"2022-10-30T02:01:14.600385Z","shell.execute_reply":"2022-10-30T02:01:15.069121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(ncols=2, figsize=(16,8))\nfor i, team in enumerate(teams):\n    xgb.plot_importance(booster=models['xgboost'][team], ax=ax[i]).set_yticklabels(list(ct.get_feature_names_out()))\n    ax[i].set_title('Feature importance for team_{}_scoring_within_10sec'.format(team))\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:08:48.396793Z","iopub.execute_input":"2022-10-30T02:08:48.397178Z","iopub.status.idle":"2022-10-30T02:08:50.437984Z","shell.execute_reply.started":"2022-10-30T02:08:48.397149Z","shell.execute_reply":"2022-10-30T02:08:50.436724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Neural Network Model","metadata":{}},{"cell_type":"code","source":"def build_and_compile_model():\n\n    n_features = X_train.shape[1]\n\n    model = Sequential()\n                        \n    model.add(Dense(units=32, activation='relu', input_shape=(n_features,)))\n    model.add(Dense(units=16, activation='relu'))\n    model.add(Dense(1, activation='sigmoid'))\n\n    model.compile(loss='binary_crossentropy',\n                  optimizer=Adam(learning_rate=0.001),\n                  metrics=BinaryCrossentropy())\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.007590Z","iopub.status.idle":"2022-10-29T16:25:20.008933Z","shell.execute_reply.started":"2022-10-29T16:25:20.008643Z","shell.execute_reply":"2022-10-29T16:25:20.008688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_loss(history):\n    # plot learning curves\n    plt.title('Learning Curves')\n    plt.xlabel('Epoch')\n    plt.ylabel('Binary Cross-entropy')\n    plt.plot(history.history['loss'], label='train')\n    plt.plot(history.history['val_loss'], label='val')\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.010746Z","iopub.status.idle":"2022-10-29T16:25:20.012140Z","shell.execute_reply.started":"2022-10-29T16:25:20.011851Z","shell.execute_reply":"2022-10-29T16:25:20.011876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stop_early = EarlyStopping(monitor='val_loss', patience=5)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.013875Z","iopub.status.idle":"2022-10-29T16:25:20.015184Z","shell.execute_reply.started":"2022-10-29T16:25:20.014912Z","shell.execute_reply":"2022-10-29T16:25:20.014938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"models['nn'] = dict()\nhistory = dict()","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.016908Z","iopub.status.idle":"2022-10-29T16:25:20.018180Z","shell.execute_reply.started":"2022-10-29T16:25:20.017910Z","shell.execute_reply":"2022-10-29T16:25:20.017935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Train the model\nfor team in teams:\n    models['nn'][team] = build_and_compile_model()\n    history[team] = models['nn'][team].fit(X_train, y_train['team_' + team + '_scoring_within_10sec'],\n                                           epochs=50, \n                                           batch_size=256, \n                                           validation_data=(X_test, y_test['team_' + team + '_scoring_within_10sec']),\n                                           callbacks = [stop_early]\n                                          )\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.019971Z","iopub.status.idle":"2022-10-29T16:25:20.021338Z","shell.execute_reply.started":"2022-10-29T16:25:20.021040Z","shell.execute_reply":"2022-10-29T16:25:20.021066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for team in teams:\n    plot_loss(history[team])","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.023059Z","iopub.status.idle":"2022-10-29T16:25:20.024442Z","shell.execute_reply.started":"2022-10-29T16:25:20.024040Z","shell.execute_reply":"2022-10-29T16:25:20.024066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model evaluation","metadata":{}},{"cell_type":"code","source":"y_pred['nn'] = dict()\ny_hat['nn'] = dict()\nfor team in teams:\n    y_hat['nn'][team] = models['nn'][team].predict(X_test)\n    y_pred['nn'][team] = np.round(y_hat['nn'][team])\n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.026249Z","iopub.status.idle":"2022-10-29T16:25:20.027528Z","shell.execute_reply.started":"2022-10-29T16:25:20.027251Z","shell.execute_reply":"2022-10-29T16:25:20.027276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_loss_score['nn'] = avg_log_loss(y_test['team_A_scoring_within_10sec'], y_hat['nn']['A'],\n                                                        y_test['team_B_scoring_within_10sec'], y_hat['nn']['B'])\nprint('Log loss score for the model: {:.4f}'.format(\n    log_loss_score['nn']))\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.029369Z","iopub.status.idle":"2022-10-29T16:25:20.030613Z","shell.execute_reply.started":"2022-10-29T16:25:20.030340Z","shell.execute_reply":"2022-10-29T16:25:20.030363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for team in teams:\n    print(f\"ROC AUC Score: {roc_auc_score(y_test['team_' + team + '_scoring_within_10sec'], y_hat['nn'][team])}\")\n    print(classification_report(y_test['team_' + team + '_scoring_within_10sec'], y_pred['nn'][team]))","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.032389Z","iopub.status.idle":"2022-10-29T16:25:20.033705Z","shell.execute_reply.started":"2022-10-29T16:25:20.033401Z","shell.execute_reply":"2022-10-29T16:25:20.033426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Principal Components Analysis","metadata":{}},{"cell_type":"markdown","source":"Before applying PCA, we must check if the feature values are normalized. We use the `ColumnTransformer` defined previously.","metadata":{}},{"cell_type":"code","source":"# Check the range of mean and standard deviation of the training dataset.\nmeans = X_train.mean(axis=1)\nstds = X_train.std(axis=1)\n\nprint('Range of means: {:.4f} to {:.4f}'.format(min(means), max(means)))\nprint('Range of standard deviations: {:.4f} to {:.4f}'.format(min(stds), max(stds)))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:09:26.193847Z","iopub.execute_input":"2022-10-30T02:09:26.194232Z","iopub.status.idle":"2022-10-30T02:09:26.268492Z","shell.execute_reply.started":"2022-10-30T02:09:26.194191Z","shell.execute_reply":"2022-10-30T02:09:26.267415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that the features are normalized. We apply PCA below.","metadata":{}},{"cell_type":"code","source":"pca = PCA()\npca.fit(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:09:28.181876Z","iopub.execute_input":"2022-10-30T02:09:28.182435Z","iopub.status.idle":"2022-10-30T02:09:28.591892Z","shell.execute_reply.started":"2022-10-30T02:09:28.182393Z","shell.execute_reply":"2022-10-30T02:09:28.590692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determine the cumulative explained variance\ncum_sum_eigenvalues = np.cumsum(pca.explained_variance_ratio_).round(6)\nexplained_variance = dict()\nfor p in np.linspace(0.86, 1, 8):\n    j = np.argwhere(cum_sum_eigenvalues >= p).flatten().min()+1\n    explained_variance[p] = [j]\nexplained_variance = pd.DataFrame(explained_variance).T\nexplained_variance.columns = ['Cumulative PC']\nexplained_variance.index.name = 'Explained Variance'\nexplained_variance\n","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:09:30.569271Z","iopub.execute_input":"2022-10-30T02:09:30.569636Z","iopub.status.idle":"2022-10-30T02:09:30.584695Z","shell.execute_reply.started":"2022-10-30T02:09:30.569604Z","shell.execute_reply":"2022-10-30T02:09:30.583708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplots(figsize=(8,3))\nplt.plot(cum_sum_eigenvalues)\nplt.title('Cumulative % of explained variance')\nplt.xlabel('Principal Component Index')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-10-30T02:09:32.005096Z","iopub.execute_input":"2022-10-30T02:09:32.005825Z","iopub.status.idle":"2022-10-30T02:09:32.250349Z","shell.execute_reply.started":"2022-10-30T02:09:32.005782Z","shell.execute_reply":"2022-10-30T02:09:32.249429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We apply the dimensionality reduction by using the first `m` eigenvectors that explain at least 90% of the variance.","metadata":{}},{"cell_type":"code","source":"explained_var_threshold = 0.9\nm = np.argwhere(cum_sum_eigenvalues >= explained_var_threshold).flatten().min()+1\npca_reduce = PCA(n_components=m)\npca_reduce.fit(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T16:25:20.049189Z","iopub.status.idle":"2022-10-29T16:25:20.050731Z","shell.execute_reply.started":"2022-10-29T16:25:20.050413Z","shell.execute_reply":"2022-10-29T16:25:20.050438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_reduced = pca_reduce.transform(X_train)\nX_test_reduced = pca_reduce.transform(X_test)\nX_train_reduced.shape, X_test_reduced.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models after dimensionality reduction\nWe fit the models again after doing PCA dimensionality reduction.","metadata":{}},{"cell_type":"markdown","source":"### Logistic regression model on reduced dataset","metadata":{}},{"cell_type":"code","source":"# %%time\n# models['logistic_regr_dim_reduce'] = dict()\n\n# models['logistic_regr_dim_reduce']['A'] = clone(models['logistic_regr']['A'])\n# models['logistic_regr_dim_reduce']['A'].fit(X_train_reduced, y_train['team_A_scoring_within_10sec'])\n\n# models['logistic_regr_dim_reduce']['B'] = clone(models['logistic_regr_dim_reduce']['A'])\n# models['logistic_regr_dim_reduce']['B'].fit(X_train_reduced, y_train['team_B_scoring_within_10sec'])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred['logistic_regr_dim_reduce'] = dict()\n# y_hat['logistic_regr_dim_reduce'] = dict()\n# for team in teams:\n#     y_pred['logistic_regr_dim_reduce'][team] = models['logistic_regr_dim_reduce'][team].predict(X_test_reduced)\n#     y_hat['logistic_regr_dim_reduce'][team] = models['logistic_regr_dim_reduce'][team].predict_proba(X_test_reduced)\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# log_loss_score['logistic_regr_dim_reduce'] = avg_log_loss(y_test['team_A_scoring_within_10sec'], y_hat['logistic_regr_dim_reduce']['A'][:, 1],\n#                                                           y_test['team_B_scoring_within_10sec'], y_hat['logistic_regr_dim_reduce']['B'][:, 1])\n# print('Log loss score for the model: {:.4f}'.format(\n#     log_loss_score['logistic_regr_dim_reduce']))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for team in teams:\n#     print(\n#         f\"ROC AUC Score: {roc_auc_score(y_test['team_' + team + '_scoring_within_10sec'], y_hat['logistic_regr_dim_reduce'][team][:,1])}\")\n#     print(classification_report(\n#         y_test['team_' + team + '_scoring_within_10sec'], y_pred['logistic_regr_dim_reduce'][team]))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost Model on reduced dataset","metadata":{}},{"cell_type":"markdown","source":"We fit the XGBoost Model again on the dataset after dimensionality reduction and evaluate its performance.","metadata":{}},{"cell_type":"code","source":"# %%time\n# models['xgboost_dim_reduce'] = dict()\n\n# models['xgboost_dim_reduce']['A'] = clone(models['xgboost']['A'])\n# models['xgboost_dim_reduce']['A'].fit(X_train_reduced, y_train['team_A_scoring_within_10sec'])\n\n\n# models['xgboost_dim_reduce']['B'] = clone(models['xgboost_dim_reduce']['A'])\n# models['xgboost_dim_reduce']['B'].fit(X_train_reduced, y_train['team_B_scoring_within_10sec'])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred['xgboost_dim_reduce'] = dict()\n# y_hat['xgboost_dim_reduce'] = dict()\n# for team in teams:\n#     y_pred['xgboost_dim_reduce'][team] = models['xgboost_dim_reduce'][team].predict(X_test_reduced)\n#     y_hat['xgboost_dim_reduce'][team] = models['xgboost_dim_reduce'][team].predict_proba(X_test_reduced)\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# log_loss_score['xgboost_dim_reduce'] = avg_log_loss(y_test['team_A_scoring_within_10sec'], y_hat['xgboost_dim_reduce']['A'][:, 1],\n#                                                         y_test['team_B_scoring_within_10sec'], y_hat['xgboost_dim_reduce']['B'][:, 1])\n# print('Log loss score for the model: {:.4f}'.format(\n#     log_loss_score['xgboost_dim_reduce']))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for team in teams:\n#     print(f\"ROC AUC Score: {roc_auc_score(y_test['team_' + team + '_scoring_within_10sec'], y_hat['xgboost_dim_reduce'][team][:,1])}\")\n#     print(classification_report(y_test['team_' + team + '_scoring_within_10sec'], y_pred['xgboost_dim_reduce'][team]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Summary","metadata":{}},{"cell_type":"code","source":"for k, v in log_loss_score.items():\n    print('Model: {}; log loss score = {:.4f}'.format(k, v))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# For Kaggle submission:","metadata":{}},{"cell_type":"code","source":"# Choose the model with the lowest log loss\nfinal_model_name = min(log_loss_score, key=log_loss_score.get)\nfinal_model = models[final_model_name]\noutput_file_name = 'submission.csv'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes_df_test = pd.read_csv(data_dir + 'test_dtypes.csv')\ndtypes_test = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\n\ndf_submission = pd.read_csv(data_dir + 'test.csv', dtype=dtypes_test)\ndf_submission = add_feature_closest_player_to_ball(df_submission)\ndf_submission = add_feature_dist_to_ball(df_submission)\ndf_submission = add_feature_ball_dist_to_net(df_submission)\nX_submission = df_submission.drop('id', axis=1)\nX_submission = ct.transform(X_submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if final_model_name == 'xgboost':\n    hat_submission_A = final_model['A'].predict_proba(X_submission)[:,1]\n    hat_submission_B = final_model['B'].predict_proba(X_submission)[:,1]\nelif final_model_name == 'nn':\n    hat_submission_A = final_model['A'].predict(X_submission)\n    hat_submission_B = final_model['B'].predict(X_submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hat_submission_A.shape, hat_submission_B.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission_file = pd.read_csv(data_dir + 'sample_submission.csv')\ndf_submission_file['team_A_scoring_within_10sec'] = hat_submission_A\ndf_submission_file['team_B_scoring_within_10sec'] = hat_submission_B\ndf_submission_file","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission_file.to_csv(output_file_name, index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}