{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🚚 Import","metadata":{}},{"cell_type":"markdown","source":"## Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np  # linear algebra\nimport pandas as pd  # data manipulation\nimport os  # file navigation\nimport gc  # garbage collection\n\n# visualization\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly import subplots\n\nfrom sklearn.model_selection import cross_validate  # k-fold Cross Validation\nfrom sklearn.preprocessing import LabelEncoder  # output binary encoding\n\nfrom xgboost import XGBClassifier  # Gradient Boosted Tree (XGBoost)\n\nfrom tensorflow.config import list_physical_devices  # check if GPU is available","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-17T13:50:07.803058Z","iopub.execute_input":"2022-10-17T13:50:07.803541Z","iopub.status.idle":"2022-10-17T13:50:14.437755Z","shell.execute_reply.started":"2022-10-17T13:50:07.803428Z","shell.execute_reply":"2022-10-17T13:50:14.436800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"# training and cross validation\nGPU = list_physical_devices('GPU') != []\nN_ESTIMATORS = 2000\nMAX_DEPTH = 8\nLEARNING_RATE = 0.01\nFOLDS = 5\n\n# data loading\nDEBUG = False\nSAMPLE = 0.2\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:50:14.439677Z","iopub.execute_input":"2022-10-17T13:50:14.440469Z","iopub.status.idle":"2022-10-17T13:50:14.520183Z","shell.execute_reply.started":"2022-10-17T13:50:14.440404Z","shell.execute_reply":"2022-10-17T13:50:14.519199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"markdown","source":"Because of memory limitations, we can't use all of the data from the 10 train .csv files.  \nInstead, we'll get a random sample of each file (for now I'm experimenting with 20-33% sample size), and combine these samples into a unique training dataset.","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-17T13:50:14.521666Z","iopub.execute_input":"2022-10-17T13:50:14.522255Z","iopub.status.idle":"2022-10-17T13:50:14.531962Z","shell.execute_reply.started":"2022-10-17T13:50:14.522219Z","shell.execute_reply":"2022-10-17T13:50:14.531089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncol_dtypes = {\n    'game_num': 'int8', 'event_id': 'int8', 'event_time': 'float16',\n    'ball_pos_x': 'float16', 'ball_pos_y': 'float16', 'ball_pos_z': 'float16',\n    'ball_vel_x': 'float16', 'ball_vel_y': 'float16', 'ball_vel_z': 'float16',\n    'p0_pos_x': 'float16', 'p0_pos_y': 'float16', 'p0_pos_z': 'float16',\n    'p0_vel_x': 'float16', 'p0_vel_y': 'float16', 'p0_vel_z': 'float16',\n    'p0_boost': 'float16', 'p1_pos_x': 'float16', 'p1_pos_y': 'float16',\n    'p1_pos_z': 'float16', 'p1_vel_x': 'float16', 'p1_vel_y': 'float16',\n    'p1_vel_z': 'float16', 'p1_boost': 'float16', 'p2_pos_x': 'float16',\n    'p2_pos_y': 'float16', 'p2_pos_z': 'float16', 'p2_vel_x': 'float16',\n    'p2_vel_y': 'float16', 'p2_vel_z': 'float16', 'p2_boost': 'float16',\n    'p3_pos_x': 'float16', 'p3_pos_y': 'float16', 'p3_pos_z': 'float16',\n    'p3_vel_x': 'float16', 'p3_vel_y': 'float16', 'p3_vel_z': 'float16',\n    'p3_boost': 'float16', 'p4_pos_x': 'float16', 'p4_pos_y': 'float16',\n    'p4_pos_z': 'float16', 'p4_vel_x': 'float16', 'p4_vel_y': 'float16',\n    'p4_vel_z': 'float16', 'p4_boost': 'float16', 'p5_pos_x': 'float16',\n    'p5_pos_y': 'float16', 'p5_pos_z': 'float16', 'p5_vel_x': 'float16',\n    'p5_vel_y': 'float16', 'p5_vel_z': 'float16', 'p5_boost': 'float16',\n    'boost0_timer': 'float16', 'boost1_timer': 'float16', 'boost2_timer': 'float16',\n    'boost3_timer': 'float16', 'boost4_timer': 'float16', 'boost5_timer': 'float16',\n    'player_scoring_next': 'O', 'team_scoring_next': 'O', 'team_A_scoring_within_10sec': 'O',\n    'team_B_scoring_within_10sec': 'O'\n}\ncols = list(col_dtypes.keys())\n\n# 20%でサンプリングする\npath_to_data = '../input/tabular-playground-series-oct-2022'\ndf = pd.DataFrame({}, columns=cols)\nfor i in range(10):\n    df_tmp = pd.read_csv(f'{path_to_data}/train_{i}.csv', dtype=col_dtypes)\n    if SAMPLE < 1:\n        df_tmp = df_tmp.sample(frac=SAMPLE, random_state=SEED)\n        \n    df = pd.concat([df, df_tmp])\n    del df_tmp\n    gc.collect()\n    if DEBUG:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:50:14.534763Z","iopub.execute_input":"2022-10-17T13:50:14.535113Z","iopub.status.idle":"2022-10-17T13:54:49.850401Z","shell.execute_reply.started":"2022-10-17T13:50:14.535079Z","shell.execute_reply":"2022-10-17T13:54:49.849491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:49.851842Z","iopub.execute_input":"2022-10-17T13:54:49.852513Z","iopub.status.idle":"2022-10-17T13:54:51.925450Z","shell.execute_reply.started":"2022-10-17T13:54:49.852475Z","shell.execute_reply":"2022-10-17T13:54:51.924477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 👀 Quick EDA","metadata":{}},{"cell_type":"code","source":"input_cols = [\n    'ball_pos_x', 'ball_pos_y', 'ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', \n    'p0_pos_x', 'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', \n    'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z',\n    'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y', 'p2_vel_z',\n    'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x', 'p3_vel_y', 'p3_vel_z',\n    'p4_pos_x', 'p4_pos_y', 'p4_pos_z', 'p4_vel_x', 'p4_vel_y', 'p4_vel_z',\n    'p5_pos_x', 'p5_pos_y', 'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z',\n    'p0_boost', 'p1_boost',  'p2_boost', 'p3_boost', 'p4_boost', 'p5_boost',\n    'boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer', 'boost4_timer', 'boost5_timer'\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:51.926790Z","iopub.execute_input":"2022-10-17T13:54:51.928701Z","iopub.status.idle":"2022-10-17T13:54:51.935382Z","shell.execute_reply.started":"2022-10-17T13:54:51.928660Z","shell.execute_reply":"2022-10-17T13:54:51.934557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_cols = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:51.936992Z","iopub.execute_input":"2022-10-17T13:54:51.937399Z","iopub.status.idle":"2022-10-17T13:54:51.945098Z","shell.execute_reply.started":"2022-10-17T13:54:51.937363Z","shell.execute_reply":"2022-10-17T13:54:51.944199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Input Variables","metadata":{}},{"cell_type":"code","source":"def int_to_grid_coord(k, n):\n    return (k // n) + 1, (k % n) + 1","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:51.948237Z","iopub.execute_input":"2022-10-17T13:54:51.948529Z","iopub.status.idle":"2022-10-17T13:54:51.954298Z","shell.execute_reply.started":"2022-10-17T13:54:51.948503Z","shell.execute_reply":"2022-10-17T13:54:51.953207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_distributions(df, row_count, col_count, title, height):\n    features = df.columns\n    fig = subplots.make_subplots(\n        rows=row_count, cols=col_count,\n        subplot_titles=features\n    )\n\n    for k, col in enumerate(features):\n        i, j = int_to_grid_coord(k, col_count)\n\n        fig.add_trace(\n            go.Histogram(\n                x=df[col].astype('float32'),\n                name=col\n            ),\n            row=i, col=j\n        )\n\n    fig.update_layout(\n        title=title,\n        height=row_count * height,\n        showlegend=False\n    )\n\n    return fig","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:51.955929Z","iopub.execute_input":"2022-10-17T13:54:51.956358Z","iopub.status.idle":"2022-10-17T13:54:51.966417Z","shell.execute_reply.started":"2022-10-17T13:54:51.956324Z","shell.execute_reply":"2022-10-17T13:54:51.965485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distributions(df[input_cols].sample(frac=0.0005), 9, 6, \"Input Variables Distributions\", 300)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:51.971872Z","iopub.execute_input":"2022-10-17T13:54:51.972159Z","iopub.status.idle":"2022-10-17T13:54:54.653208Z","shell.execute_reply.started":"2022-10-17T13:54:51.972134Z","shell.execute_reply":"2022-10-17T13:54:54.652282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Output Variables\n10秒以内に得点できているのはごく一部","metadata":{}},{"cell_type":"code","source":"plot_distributions(df[output_cols].sample(frac=0.0005), 1, 2, \"Output Variables Distributions\", height=600)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:54.654163Z","iopub.execute_input":"2022-10-17T13:54:54.660373Z","iopub.status.idle":"2022-10-17T13:54:54.936044Z","shell.execute_reply.started":"2022-10-17T13:54:54.660317Z","shell.execute_reply":"2022-10-17T13:54:54.934992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"Shoutout [this post by samuelcortinhas](https://www.kaggle.com/competitions/tabular-playground-series-oct-2022/discussion/356852) for the idea.","metadata":{}},{"cell_type":"markdown","source":"Let's derive 2 new features:\n* For each player (and the ball) let's get their velocity's magnitude.\n* For each player, let's get their distance from the ball.","metadata":{}},{"cell_type":"markdown","source":"## Euclidian Norm","metadata":{}},{"cell_type":"markdown","source":"For these 2 new features, we'll need to get the 3D euclidian norm of a vector:\n$$ \\| \\overrightarrow{v} \\| = \\sqrt{x^2 + y^2 + z^2} $$\nWe'll use numpy's linalg.norm() method for that.","metadata":{}},{"cell_type":"code","source":"def euclidian_norm(x):\n    return np.linalg.norm(x, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:54.937754Z","iopub.execute_input":"2022-10-17T13:54:54.938130Z","iopub.status.idle":"2022-10-17T13:54:54.943205Z","shell.execute_reply.started":"2022-10-17T13:54:54.938094Z","shell.execute_reply":"2022-10-17T13:54:54.942197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's group the x, y and z variables by player and ball categories\n# this simplifies the code for the euclidian norm calculation\nvel_groups = {\n    f\"{el}_vel\": [f'{el}_vel_x', f'{el}_vel_y', f'{el}_vel_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups = {\n    f\"{el}_pos\": [f'{el}_pos_x', f'{el}_pos_y', f'{el}_pos_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:54.944805Z","iopub.execute_input":"2022-10-17T13:54:54.945527Z","iopub.status.idle":"2022-10-17T13:54:54.956632Z","shell.execute_reply.started":"2022-10-17T13:54:54.945485Z","shell.execute_reply":"2022-10-17T13:54:54.955600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# velocity magnitude\nfor col, vec in vel_groups.items():\n    df[col] = euclidian_norm(df[vec])","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:54:54.958048Z","iopub.execute_input":"2022-10-17T13:54:54.958459Z","iopub.status.idle":"2022-10-17T13:55:00.936707Z","shell.execute_reply.started":"2022-10-17T13:54:54.958402Z","shell.execute_reply":"2022-10-17T13:55:00.935686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distance from ball\nfor col, vec in pos_groups.items():\n    df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:00.941266Z","iopub.execute_input":"2022-10-17T13:55:00.942044Z","iopub.status.idle":"2022-10-17T13:55:07.345913Z","shell.execute_reply.started":"2022-10-17T13:55:00.942005Z","shell.execute_reply":"2022-10-17T13:55:07.344889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧹 Cleaning","metadata":{}},{"cell_type":"markdown","source":"We drop the columns below because they should not influence the results.  \ngame_num, event_id and event_time are irrelevant.  \nplayer_scoring_next and team_scoring next are a form of data leakage, as in they're synonymous with the output variable.  \nball_pos_ball_dist is always 0, the distance between the ball and itself.","metadata":{}},{"cell_type":"markdown","source":"## Dropping columns","metadata":{}},{"cell_type":"code","source":"cols_to_drop = [\n    'game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'ball_pos_ball_dist'\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:07.347317Z","iopub.execute_input":"2022-10-17T13:55:07.348158Z","iopub.status.idle":"2022-10-17T13:55:07.354056Z","shell.execute_reply.started":"2022-10-17T13:55:07.348119Z","shell.execute_reply":"2022-10-17T13:55:07.352934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(columns=cols_to_drop)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:07.355409Z","iopub.execute_input":"2022-10-17T13:55:07.356460Z","iopub.status.idle":"2022-10-17T13:55:08.813689Z","shell.execute_reply.started":"2022-10-17T13:55:07.356398Z","shell.execute_reply":"2022-10-17T13:55:08.812701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:08.815251Z","iopub.execute_input":"2022-10-17T13:55:08.815662Z","iopub.status.idle":"2022-10-17T13:55:09.572050Z","shell.execute_reply.started":"2022-10-17T13:55:08.815623Z","shell.execute_reply":"2022-10-17T13:55:09.570920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dropping rows containing NaN","metadata":{}},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:09.573689Z","iopub.execute_input":"2022-10-17T13:55:09.574090Z","iopub.status.idle":"2022-10-17T13:55:10.434527Z","shell.execute_reply.started":"2022-10-17T13:55:09.574053Z","shell.execute_reply":"2022-10-17T13:55:10.433338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"例えば、p0_pos_xでの欠損値について","metadata":{}},{"cell_type":"code","source":"null_p0_pos_x_count = df['p0_pos_x'].isna().sum()\nnull_p0_pos_x_perc = null_p0_pos_x_count / df.shape[0]\nprint(f\"Missing {null_p0_pos_x_count} values ({null_p0_pos_x_perc:.2%})\")","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:10.435916Z","iopub.execute_input":"2022-10-17T13:55:10.436555Z","iopub.status.idle":"2022-10-17T13:55:10.456763Z","shell.execute_reply.started":"2022-10-17T13:55:10.436515Z","shell.execute_reply":"2022-10-17T13:55:10.455745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To keep things simple, let's just drop all null values.","metadata":{}},{"cell_type":"markdown","source":"欠損値のある、\"行\"を削除","metadata":{}},{"cell_type":"code","source":"df = df.dropna(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:10.458115Z","iopub.execute_input":"2022-10-17T13:55:10.458559Z","iopub.status.idle":"2022-10-17T13:55:14.515865Z","shell.execute_reply.started":"2022-10-17T13:55:10.458521Z","shell.execute_reply":"2022-10-17T13:55:14.514813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"これで、欠損値を含む列がなくなる  \n以下で確認する","metadata":{}},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:14.517394Z","iopub.execute_input":"2022-10-17T13:55:14.517992Z","iopub.status.idle":"2022-10-17T13:55:15.314411Z","shell.execute_reply.started":"2022-10-17T13:55:14.517953Z","shell.execute_reply":"2022-10-17T13:55:15.313217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"これが最終版のトレインデータ","metadata":{}},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:15.315800Z","iopub.execute_input":"2022-10-17T13:55:15.316424Z","iopub.status.idle":"2022-10-17T13:55:16.011331Z","shell.execute_reply.started":"2022-10-17T13:55:15.316384Z","shell.execute_reply":"2022-10-17T13:55:16.010368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:16.012878Z","iopub.execute_input":"2022-10-17T13:55:16.013251Z","iopub.status.idle":"2022-10-17T13:55:16.022266Z","shell.execute_reply.started":"2022-10-17T13:55:16.013215Z","shell.execute_reply":"2022-10-17T13:55:16.020549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🚀 Model Training","metadata":{}},{"cell_type":"code","source":"X = df.drop([\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"], axis=1)\nX","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:16.024001Z","iopub.execute_input":"2022-10-17T13:55:16.024533Z","iopub.status.idle":"2022-10-17T13:55:17.097480Z","shell.execute_reply.started":"2022-10-17T13:55:16.024496Z","shell.execute_reply":"2022-10-17T13:55:17.096462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df[[\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"]]\ny = y.astype(int)\ny","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:17.100028Z","iopub.execute_input":"2022-10-17T13:55:17.100732Z","iopub.status.idle":"2022-10-17T13:55:17.847448Z","shell.execute_reply.started":"2022-10-17T13:55:17.100692Z","shell.execute_reply":"2022-10-17T13:55:17.846344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import log_loss, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:17.849052Z","iopub.execute_input":"2022-10-17T13:55:17.849428Z","iopub.status.idle":"2022-10-17T13:55:18.088172Z","shell.execute_reply.started":"2022-10-17T13:55:17.849393Z","shell.execute_reply":"2022-10-17T13:55:18.087259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits = 5\nseed = 42\n\nparams = {'objective':'binary',\n          'metric' : 'auc',\n          'seed': 42,\n          'num_leaves' : 64,\n          'min_child_samples': 20,\n          'max_depth' : 6,\n          'n_estimators': 300,\n          'learning_rate': 0.1,\n         }\n\nmodel_A = lgb.LGBMClassifier(**params)   \nmodel_B = lgb.LGBMClassifier(**params)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:18.093343Z","iopub.execute_input":"2022-10-17T13:55:18.093650Z","iopub.status.idle":"2022-10-17T13:55:18.099516Z","shell.execute_reply.started":"2022-10-17T13:55:18.093624Z","shell.execute_reply":"2022-10-17T13:55:18.098608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_A, X_val_A = train_test_split(X,test_size = 0.2,random_state = seed)\ny_train_A, y_val_A = train_test_split(y[\"team_A_scoring_within_10sec\"],test_size = 0.2,random_state = seed)\nX_train_B, X_val_B = train_test_split(X,test_size = 0.2,random_state = seed)\ny_train_B, y_val_B = train_test_split(y[\"team_B_scoring_within_10sec\"],test_size = 0.2,random_state = seed)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:18.100834Z","iopub.execute_input":"2022-10-17T13:55:18.101796Z","iopub.status.idle":"2022-10-17T13:55:26.987521Z","shell.execute_reply.started":"2022-10-17T13:55:18.101754Z","shell.execute_reply":"2022-10-17T13:55:26.986489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model A","metadata":{}},{"cell_type":"code","source":"model_A.fit(X_train_A,y_train_A)\npred_ = model_A.predict_proba(X_val_A)[:,1]\nloss_A = log_loss(y_val_A ,pred_)\nloss_A","metadata":{"execution":{"iopub.status.busy":"2022-10-17T13:55:26.989192Z","iopub.execute_input":"2022-10-17T13:55:26.989591Z","iopub.status.idle":"2022-10-17T14:02:59.818740Z","shell.execute_reply.started":"2022-10-17T13:55:26.989554Z","shell.execute_reply":"2022-10-17T14:02:59.815883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model B","metadata":{}},{"cell_type":"code","source":"model_B.fit(X_train_B,y_train_B)\npred_ = model_B.predict_proba(X_val_B)[:,1]\nloss_B = log_loss(y_val_B ,pred_)\nloss_B","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.820117Z","iopub.status.idle":"2022-10-17T14:02:59.820623Z","shell.execute_reply.started":"2022-10-17T14:02:59.820357Z","shell.execute_reply":"2022-10-17T14:02:59.820381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\ndf_test_ = df_test.copy()\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.822031Z","iopub.status.idle":"2022-10-17T14:02:59.822943Z","shell.execute_reply.started":"2022-10-17T14:02:59.822656Z","shell.execute_reply":"2022-10-17T14:02:59.822685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(df):\n    # velocity magnitude\n    for col, vec in vel_groups.items():\n        df[col] = euclidian_norm(df[vec])\n    \n    # ball distance\n    for col, vec in pos_groups.items():\n        df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)\n    \n    df = df.drop(columns=['ball_pos_ball_dist'])\n    df = df.drop([\"id\"], axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.824687Z","iopub.status.idle":"2022-10-17T14:02:59.825159Z","shell.execute_reply.started":"2022-10-17T14:02:59.824920Z","shell.execute_reply":"2022-10-17T14:02:59.824942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = preprocess(df_test)\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.826814Z","iopub.status.idle":"2022-10-17T14:02:59.827296Z","shell.execute_reply.started":"2022-10-17T14:02:59.827050Z","shell.execute_reply":"2022-10-17T14:02:59.827073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"prediction","metadata":{}},{"cell_type":"code","source":"pred_A = model_A.predict_proba(df_test)[:,1]\npred_B = model_B.predict_proba(df_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.828907Z","iopub.status.idle":"2022-10-17T14:02:59.829366Z","shell.execute_reply.started":"2022-10-17T14:02:59.829128Z","shell.execute_reply":"2022-10-17T14:02:59.829149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💌 Submission","metadata":{}},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": df_test_['id'],\n        \"team_A_scoring_within_10sec\": pred_A,\n        \"team_B_scoring_within_10sec\": pred_B\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.831018Z","iopub.status.idle":"2022-10-17T14:02:59.831502Z","shell.execute_reply.started":"2022-10-17T14:02:59.831241Z","shell.execute_reply":"2022-10-17T14:02:59.831265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:02:59.833145Z","iopub.status.idle":"2022-10-17T14:02:59.833925Z","shell.execute_reply.started":"2022-10-17T14:02:59.833665Z","shell.execute_reply":"2022-10-17T14:02:59.833688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T14:20:38.708260Z","iopub.execute_input":"2022-10-17T14:20:38.708642Z","iopub.status.idle":"2022-10-17T14:20:38.731201Z","shell.execute_reply.started":"2022-10-17T14:20:38.708610Z","shell.execute_reply":"2022-10-17T14:20:38.729873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}