{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I referred [chazzer](https://www.kaggle.com/chazzer)'s [notebook](https://www.kaggle.com/code/chazzer/rocket-league-xgboost-feat-engineering-cv) and [my previous one](https://www.kaggle.com/code/shoooono/oct2022-feature-engineering-lightbgm).","metadata":{}},{"cell_type":"markdown","source":"# 🚚 Import","metadata":{}},{"cell_type":"markdown","source":"## Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np  # linear algebra\nimport pandas as pd  # data manipulation\nimport os  # file navigation\nimport gc  # garbage collection\n\n# visualization\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly import subplots\n\nfrom sklearn.model_selection import cross_validate  # k-fold Cross Validation\nfrom sklearn.preprocessing import LabelEncoder  # output binary encoding\n\n# from xgboost import XGBClassifier  # Gradient Boosted Tree (XGBoost)\n\nfrom tensorflow.config import list_physical_devices  # check if GPU is available","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-24T13:55:00.246294Z","iopub.execute_input":"2022-10-24T13:55:00.247121Z","iopub.status.idle":"2022-10-24T13:55:00.260268Z","shell.execute_reply.started":"2022-10-24T13:55:00.247075Z","shell.execute_reply":"2022-10-24T13:55:00.259124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"# training and cross validation\nGPU = list_physical_devices('GPU') != []\nN_ESTIMATORS  = 2000\nMAX_DEPTH     = 8\nLEARNING_RATE = 0.01\nFOLDS         = 5\n\n# data loading\nDEBUG   = False\nSAMPLE  = 0.3\nSEED    = 42","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:55:00.263482Z","iopub.execute_input":"2022-10-24T13:55:00.266228Z","iopub.status.idle":"2022-10-24T13:55:00.274962Z","shell.execute_reply.started":"2022-10-24T13:55:00.266178Z","shell.execute_reply":"2022-10-24T13:55:00.273985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"markdown","source":"Because of memory limitations, we can't use all of the data from the 10 train .csv files.  \nInstead, we'll get a random sample of each file (for now I'm experimenting with 20-33% sample size), and combine these samples into a unique training dataset.","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-24T13:55:00.276620Z","iopub.execute_input":"2022-10-24T13:55:00.277479Z","iopub.status.idle":"2022-10-24T13:55:00.289167Z","shell.execute_reply.started":"2022-10-24T13:55:00.277444Z","shell.execute_reply":"2022-10-24T13:55:00.288249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncol_dtypes = {\n    'game_num': 'int8', 'event_id': 'int8', 'event_time': 'float16',\n    'ball_pos_x': 'float16', 'ball_pos_y': 'float16', 'ball_pos_z': 'float16',\n    'ball_vel_x': 'float16', 'ball_vel_y': 'float16', 'ball_vel_z': 'float16',\n    'p0_pos_x': 'float16', 'p0_pos_y': 'float16', 'p0_pos_z': 'float16',\n    'p0_vel_x': 'float16', 'p0_vel_y': 'float16', 'p0_vel_z': 'float16',\n    'p0_boost': 'float16', 'p1_pos_x': 'float16', 'p1_pos_y': 'float16',\n    'p1_pos_z': 'float16', 'p1_vel_x': 'float16', 'p1_vel_y': 'float16',\n    'p1_vel_z': 'float16', 'p1_boost': 'float16', 'p2_pos_x': 'float16',\n    'p2_pos_y': 'float16', 'p2_pos_z': 'float16', 'p2_vel_x': 'float16',\n    'p2_vel_y': 'float16', 'p2_vel_z': 'float16', 'p2_boost': 'float16',\n    'p3_pos_x': 'float16', 'p3_pos_y': 'float16', 'p3_pos_z': 'float16',\n    'p3_vel_x': 'float16', 'p3_vel_y': 'float16', 'p3_vel_z': 'float16',\n    'p3_boost': 'float16', 'p4_pos_x': 'float16', 'p4_pos_y': 'float16',\n    'p4_pos_z': 'float16', 'p4_vel_x': 'float16', 'p4_vel_y': 'float16',\n    'p4_vel_z': 'float16', 'p4_boost': 'float16', 'p5_pos_x': 'float16',\n    'p5_pos_y': 'float16', 'p5_pos_z': 'float16', 'p5_vel_x': 'float16',\n    'p5_vel_y': 'float16', 'p5_vel_z': 'float16', 'p5_boost': 'float16',\n    'boost0_timer': 'float16', 'boost1_timer': 'float16', 'boost2_timer': 'float16',\n    'boost3_timer': 'float16', 'boost4_timer': 'float16', 'boost5_timer': 'float16',\n    'player_scoring_next': 'O', 'team_scoring_next': 'O', 'team_A_scoring_within_10sec': 'O',\n    'team_B_scoring_within_10sec': 'O'\n}\ncols = list(col_dtypes.keys())\n\n# read csv and sampling\npath_to_data = '../input/tabular-playground-series-oct-2022'\ndf = pd.DataFrame({}, columns=cols)\nfor i in range(10):\n    df_tmp = pd.read_csv(f'{path_to_data}/train_{i}.csv', dtype=col_dtypes)\n    if SAMPLE < 1:\n        df_tmp = df_tmp.sample(frac=SAMPLE, random_state=SEED)\n        \n    df = pd.concat([df, df_tmp])\n    del df_tmp\n    gc.collect()\n    if DEBUG:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:55:00.290724Z","iopub.execute_input":"2022-10-24T13:55:00.291361Z","iopub.status.idle":"2022-10-24T13:57:50.201270Z","shell.execute_reply.started":"2022-10-24T13:55:00.291327Z","shell.execute_reply":"2022-10-24T13:57:50.200233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:50.204456Z","iopub.execute_input":"2022-10-24T13:57:50.204746Z","iopub.status.idle":"2022-10-24T13:57:50.632779Z","shell.execute_reply.started":"2022-10-24T13:57:50.204718Z","shell.execute_reply":"2022-10-24T13:57:50.631685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"Let's derive 4 new features:\n* For each player (and the ball)  velocity's magnitude.\n* For each player,  distance from the ball.\n* For each player, (and the ball) distance from goal.\n* For each player, cosine betweem vectors, from player to goal, and from player to ball. ","metadata":{}},{"cell_type":"markdown","source":"## Euclidian Norm","metadata":{}},{"cell_type":"markdown","source":"We'll need to get the 3D euclidian norm of a vector:\n$$ \\| \\overrightarrow{v} \\| = \\sqrt{x^2 + y^2 + z^2} $$\nWe'll use numpy's linalg.norm() method for that.","metadata":{}},{"cell_type":"code","source":"def euclidian_norm(x):\n    return np.linalg.norm(x, axis=1)#行の３成分(x,y,z)で絶対値取る","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:50.634564Z","iopub.execute_input":"2022-10-24T13:57:50.634955Z","iopub.status.idle":"2022-10-24T13:57:50.640891Z","shell.execute_reply.started":"2022-10-24T13:57:50.634918Z","shell.execute_reply":"2022-10-24T13:57:50.639914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vel_groups = {\n    f\"{el}_vel\": [f'{el}_vel_x', f'{el}_vel_y', f'{el}_vel_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups = {\n    f\"{el}_pos\": [f'{el}_pos_x', f'{el}_pos_y', f'{el}_pos_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:50.642521Z","iopub.execute_input":"2022-10-24T13:57:50.643169Z","iopub.status.idle":"2022-10-24T13:57:50.650958Z","shell.execute_reply.started":"2022-10-24T13:57:50.643133Z","shell.execute_reply":"2022-10-24T13:57:50.649861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Velocity magnitude","metadata":{}},{"cell_type":"code","source":"for col, vec in vel_groups.items():\n    df[col] = euclidian_norm(df[vec])","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:50.652284Z","iopub.execute_input":"2022-10-24T13:57:50.653172Z","iopub.status.idle":"2022-10-24T13:57:52.239776Z","shell.execute_reply.started":"2022-10-24T13:57:50.653134Z","shell.execute_reply":"2022-10-24T13:57:52.238807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distance from ball","metadata":{}},{"cell_type":"code","source":"for col, vec in pos_groups.items():\n    df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:52.241347Z","iopub.execute_input":"2022-10-24T13:57:52.241921Z","iopub.status.idle":"2022-10-24T13:57:53.859081Z","shell.execute_reply.started":"2022-10-24T13:57:52.241879Z","shell.execute_reply":"2022-10-24T13:57:53.858086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dropping columns\n\nWe drop the columns below because they should not influence the results.  \ngame_num, event_id and event_time are irrelevant.  \nplayer_scoring_next and team_scoring next are a form of data leakage, as in they're synonymous with the output variable.  \nball_pos_ball_dist is always 0, the distance between the ball and itself.","metadata":{}},{"cell_type":"code","source":"cols_to_drop = [\n    'game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'ball_pos_ball_dist'\n]\ndf = df.drop(columns=cols_to_drop)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:53.860540Z","iopub.execute_input":"2022-10-24T13:57:53.860896Z","iopub.status.idle":"2022-10-24T13:57:54.260068Z","shell.execute_reply.started":"2022-10-24T13:57:53.860860Z","shell.execute_reply":"2022-10-24T13:57:54.258989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:54.261586Z","iopub.execute_input":"2022-10-24T13:57:54.262591Z","iopub.status.idle":"2022-10-24T13:57:54.612583Z","shell.execute_reply.started":"2022-10-24T13:57:54.262546Z","shell.execute_reply":"2022-10-24T13:57:54.611408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distance from goal","metadata":{}},{"cell_type":"code","source":"goal_a = np.array( [0, -120, 1.2], dtype='float16')\ngoal_b = np.array( [0,  120, 1.2], dtype='float16')","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:54.614087Z","iopub.execute_input":"2022-10-24T13:57:54.614595Z","iopub.status.idle":"2022-10-24T13:57:54.620757Z","shell.execute_reply.started":"2022-10-24T13:57:54.614551Z","shell.execute_reply":"2022-10-24T13:57:54.619593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vec in pos_groups.items():\n    df[col + \"_goal_a_dist\"] = euclidian_norm(df[vec].values - goal_a)\n    df[col + \"_goal_b_dist\"] = euclidian_norm(df[vec].values - goal_b)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:54.622457Z","iopub.execute_input":"2022-10-24T13:57:54.623157Z","iopub.status.idle":"2022-10-24T13:57:58.104460Z","shell.execute_reply.started":"2022-10-24T13:57:54.623117Z","shell.execute_reply":"2022-10-24T13:57:58.103419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:58.109652Z","iopub.execute_input":"2022-10-24T13:57:58.109953Z","iopub.status.idle":"2022-10-24T13:57:58.497338Z","shell.execute_reply.started":"2022-10-24T13:57:58.109925Z","shell.execute_reply":"2022-10-24T13:57:58.496051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cos","metadata":{}},{"cell_type":"code","source":"# test\n# vec                    = pos_groups[\"p0_pos\"]\n# vec_a                  = goal_a - df[vec].values                                # P0からゴールのベクトル\n# vec_c                  = df[pos_groups[\"ball_pos\"]].values  - df[vec].values    # P0からボールのベクトル\n# df[\"p0_pos\"+\"_inner_a\"]  = np.sum(vec_a*vec_c,axis=1)\n# df[\"p0_pos\"+\"_cos_a\"]  = np.sum(vec_a*vec_c,axis=1)/euclidian_norm(vec_a)/euclidian_norm(vec_c)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:58.499032Z","iopub.execute_input":"2022-10-24T13:57:58.499690Z","iopub.status.idle":"2022-10-24T13:57:58.504858Z","shell.execute_reply.started":"2022-10-24T13:57:58.499647Z","shell.execute_reply":"2022-10-24T13:57:58.503559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vec in pos_groups.items():\n    vec_a                  = goal_a - df[vec].values                              # PからゴールAのベクトル\n    vec_c                  = df[pos_groups[\"ball_pos\"]].values  - df[vec].values  # Pからボールのベクトル\n    df[col + \"_cos_a\"]   = np.sum(vec_a*vec_c,axis=1)/euclidian_norm(vec_a)/euclidian_norm(vec_c)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:57:58.506693Z","iopub.execute_input":"2022-10-24T13:57:58.507269Z","iopub.status.idle":"2022-10-24T13:58:02.037321Z","shell.execute_reply.started":"2022-10-24T13:57:58.507229Z","shell.execute_reply":"2022-10-24T13:58:02.036322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col, vec in pos_groups.items():\n    vec_a                  = goal_b - df[vec].values                              # PからゴールBのベクトル\n    vec_c                  = df[pos_groups[\"ball_pos\"]].values  - df[vec].values  # Pからボールのベクトル\n    df[col + \"_cos_b\"]   = np.sum(vec_a*vec_c,axis=1)/euclidian_norm(vec_a)/euclidian_norm(vec_c)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:02.038937Z","iopub.execute_input":"2022-10-24T13:58:02.039358Z","iopub.status.idle":"2022-10-24T13:58:05.669726Z","shell.execute_reply.started":"2022-10-24T13:58:02.039318Z","shell.execute_reply":"2022-10-24T13:58:05.668654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:05.671229Z","iopub.execute_input":"2022-10-24T13:58:05.672372Z","iopub.status.idle":"2022-10-24T13:58:06.126414Z","shell.execute_reply.started":"2022-10-24T13:58:05.672328Z","shell.execute_reply":"2022-10-24T13:58:06.124355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Drop ball_pos_inner","metadata":{}},{"cell_type":"code","source":"cols_to_drop = [\n    'ball_pos_cos_a', 'ball_pos_cos_b'\n]\ndf = df.drop(columns=cols_to_drop)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:06.127838Z","iopub.execute_input":"2022-10-24T13:58:06.128314Z","iopub.status.idle":"2022-10-24T13:58:06.645813Z","shell.execute_reply.started":"2022-10-24T13:58:06.128272Z","shell.execute_reply":"2022-10-24T13:58:06.644717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dropping rows containing NaN","metadata":{}},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:06.647369Z","iopub.execute_input":"2022-10-24T13:58:06.647760Z","iopub.status.idle":"2022-10-24T13:58:06.965562Z","shell.execute_reply.started":"2022-10-24T13:58:06.647719Z","shell.execute_reply":"2022-10-24T13:58:06.964565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For example, \"p0_pos_x\"","metadata":{}},{"cell_type":"code","source":"null_p0_pos_x_count = df['p0_pos_x'].isna().sum()\nnull_p0_pos_x_perc = null_p0_pos_x_count / df.shape[0]\nprint(f\"Missing {null_p0_pos_x_count} values ({null_p0_pos_x_perc:.2%})\")","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:06.969964Z","iopub.execute_input":"2022-10-24T13:58:06.970981Z","iopub.status.idle":"2022-10-24T13:58:06.990535Z","shell.execute_reply.started":"2022-10-24T13:58:06.970935Z","shell.execute_reply":"2022-10-24T13:58:06.989320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"drop rows containg num","metadata":{}},{"cell_type":"code","source":"df = df.dropna(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:06.992525Z","iopub.execute_input":"2022-10-24T13:58:06.993368Z","iopub.status.idle":"2022-10-24T13:58:08.414458Z","shell.execute_reply.started":"2022-10-24T13:58:06.993322Z","shell.execute_reply":"2022-10-24T13:58:08.413408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"check num","metadata":{}},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:08.416047Z","iopub.execute_input":"2022-10-24T13:58:08.416411Z","iopub.status.idle":"2022-10-24T13:58:08.699050Z","shell.execute_reply.started":"2022-10-24T13:58:08.416374Z","shell.execute_reply":"2022-10-24T13:58:08.698035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This is the final train dataframe","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:07:02.074078Z","iopub.execute_input":"2022-10-24T14:07:02.074753Z","iopub.status.idle":"2022-10-24T14:07:02.457394Z","shell.execute_reply.started":"2022-10-24T14:07:02.074713Z","shell.execute_reply":"2022-10-24T14:07:02.456236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🚀 Model Training","metadata":{}},{"cell_type":"code","source":"X = df.drop([\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:08.793849Z","iopub.execute_input":"2022-10-24T13:58:08.794311Z","iopub.status.idle":"2022-10-24T13:58:09.117467Z","shell.execute_reply.started":"2022-10-24T13:58:08.794271Z","shell.execute_reply":"2022-10-24T13:58:09.116375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df[[\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"]]\ny = y.astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:09.118927Z","iopub.execute_input":"2022-10-24T13:58:09.119429Z","iopub.status.idle":"2022-10-24T13:58:09.296574Z","shell.execute_reply.started":"2022-10-24T13:58:09.119389Z","shell.execute_reply":"2022-10-24T13:58:09.295475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import log_loss, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:09.298063Z","iopub.execute_input":"2022-10-24T13:58:09.298539Z","iopub.status.idle":"2022-10-24T13:58:09.507335Z","shell.execute_reply.started":"2022-10-24T13:58:09.298502Z","shell.execute_reply":"2022-10-24T13:58:09.506410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits = 5\nseed = 42\n\nparams = {'objective':'binary',\n          'metric' : 'auc',\n          'seed': 42,\n          'num_leaves' : 64,\n          'min_child_samples': 20,\n          'max_depth' : 6,\n          'n_estimators': 300,\n          'learning_rate': 0.1,\n         }\n\nmodel_A = lgb.LGBMClassifier(**params)   \nmodel_B = lgb.LGBMClassifier(**params)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:09.508899Z","iopub.execute_input":"2022-10-24T13:58:09.509334Z","iopub.status.idle":"2022-10-24T13:58:09.517769Z","shell.execute_reply.started":"2022-10-24T13:58:09.509295Z","shell.execute_reply":"2022-10-24T13:58:09.516756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_A, X_val_A = train_test_split(X,test_size = 0.2,random_state = seed)\ny_train_A, y_val_A = train_test_split(y[\"team_A_scoring_within_10sec\"],test_size = 0.2,random_state = seed)\nX_train_B, X_val_B = train_test_split(X,test_size = 0.2,random_state = seed)\ny_train_B, y_val_B = train_test_split(y[\"team_B_scoring_within_10sec\"],test_size = 0.2,random_state = seed)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:09.519105Z","iopub.execute_input":"2022-10-24T13:58:09.520069Z","iopub.status.idle":"2022-10-24T13:58:12.071437Z","shell.execute_reply.started":"2022-10-24T13:58:09.520031Z","shell.execute_reply":"2022-10-24T13:58:12.070376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model A","metadata":{}},{"cell_type":"code","source":"%%time\nmodel_A.fit(X_train_A,y_train_A)\npred_ = model_A.predict_proba(X_val_A)[:,1]\nloss_A = log_loss(y_val_A ,pred_)\nloss_A","metadata":{"execution":{"iopub.status.busy":"2022-10-24T13:58:12.072825Z","iopub.execute_input":"2022-10-24T13:58:12.073681Z","iopub.status.idle":"2022-10-24T14:00:39.011562Z","shell.execute_reply.started":"2022-10-24T13:58:12.073640Z","shell.execute_reply":"2022-10-24T14:00:39.010560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# model B","metadata":{}},{"cell_type":"code","source":"%%time\nmodel_B.fit(X_train_B,y_train_B)\npred_ = model_B.predict_proba(X_val_B)[:,1]\nloss_B = log_loss(y_val_B ,pred_)\nloss_B","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:00:39.013137Z","iopub.execute_input":"2022-10-24T14:00:39.013829Z","iopub.status.idle":"2022-10-24T14:03:08.700438Z","shell.execute_reply.started":"2022-10-24T14:00:39.013789Z","shell.execute_reply":"2022-10-24T14:03:08.699294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test data","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\ndf_test_ = df_test.copy()\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:04:48.260074Z","iopub.execute_input":"2022-10-24T14:04:48.260454Z","iopub.status.idle":"2022-10-24T14:04:53.073710Z","shell.execute_reply.started":"2022-10-24T14:04:48.260424Z","shell.execute_reply":"2022-10-24T14:04:53.072558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(df):\n    # velocity magnitude\n    for col, vec in vel_groups.items():\n        df[col] = euclidian_norm(df[vec])\n    \n    # ball distance\n    for col, vec in pos_groups.items():\n        df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)\n        \n    # goal distance    \n    for col, vec in pos_groups.items():\n        df[col + \"_goal_a_dist\"] = euclidian_norm(df[vec].values - goal_a)\n        df[col + \"_goal_b_dist\"] = euclidian_norm(df[vec].values - goal_b)\n        \n    # cos goalA\n    for col, vec in pos_groups.items():\n        vec_a                  = goal_a - df[vec].values                              # PからゴールAのベクトル\n        vec_c                  = df[pos_groups[\"ball_pos\"]].values  - df[vec].values  # Pからボールのベクトル\n        df[col + \"_cos_a\"]   = np.sum(vec_a*vec_c,axis=1)/euclidian_norm(vec_a)/euclidian_norm(vec_c)\n\n    # cos goalB\n    for col, vec in pos_groups.items():\n        vec_a                  = goal_b - df[vec].values                              # PからゴールBのベクトル\n        vec_c                  = df[pos_groups[\"ball_pos\"]].values  - df[vec].values  # Pからボールのベクトル\n        df[col + \"_cos_b\"]   = np.sum(vec_a*vec_c,axis=1)/euclidian_norm(vec_a)/euclidian_norm(vec_c)\n    \n        \n    cols_to_drop = [\n    'ball_pos_cos_a', 'ball_pos_cos_b', 'ball_pos_ball_dist'\n    ]\n    df = df.drop(columns= cols_to_drop)\n    df = df.drop([\"id\"], axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:04:56.392491Z","iopub.execute_input":"2022-10-24T14:04:56.392859Z","iopub.status.idle":"2022-10-24T14:04:56.404547Z","shell.execute_reply.started":"2022-10-24T14:04:56.392827Z","shell.execute_reply":"2022-10-24T14:04:56.403212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = preprocess(df_test)\ndf_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:04:58.744749Z","iopub.execute_input":"2022-10-24T14:04:58.745120Z","iopub.status.idle":"2022-10-24T14:05:16.939471Z","shell.execute_reply.started":"2022-10-24T14:04:58.745088Z","shell.execute_reply":"2022-10-24T14:05:16.937388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# prediction","metadata":{}},{"cell_type":"code","source":"pred_A = model_A.predict_proba(df_test)[:,1]\npred_B = model_B.predict_proba(df_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:03:33.827621Z","iopub.execute_input":"2022-10-24T14:03:33.828633Z","iopub.status.idle":"2022-10-24T14:03:57.301629Z","shell.execute_reply.started":"2022-10-24T14:03:33.828595Z","shell.execute_reply":"2022-10-24T14:03:57.300495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💌 Submission","metadata":{}},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": df_test_['id'],\n        \"team_A_scoring_within_10sec\": pred_A,\n        \"team_B_scoring_within_10sec\": pred_B\n    }\n)\ndf_submission","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:03:57.303433Z","iopub.execute_input":"2022-10-24T14:03:57.304056Z","iopub.status.idle":"2022-10-24T14:03:57.324380Z","shell.execute_reply.started":"2022-10-24T14:03:57.304014Z","shell.execute_reply":"2022-10-24T14:03:57.323184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-24T14:03:57.325962Z","iopub.execute_input":"2022-10-24T14:03:57.326455Z","iopub.status.idle":"2022-10-24T14:03:59.782519Z","shell.execute_reply.started":"2022-10-24T14:03:57.326414Z","shell.execute_reply":"2022-10-24T14:03:59.781462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}