{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Fine tuning of the solution proposed by [chazzer](https://www.kaggle.com/code/chazzer/rocket-league-xgboost-feat-engineering-cv/notebook)","metadata":{}},{"cell_type":"code","source":"import numpy as np  # linear algebra\nimport pandas as pd  # data manipulation\nimport os  # file navigation\nimport gc  # garbage collection\n\n# visualization\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly import subplots\n\nfrom sklearn.model_selection import cross_validate  # k-fold Cross Validation\nfrom sklearn.preprocessing import LabelEncoder  # output binary encoding\n\nfrom xgboost import XGBClassifier  # Gradient Boosted Tree (XGBoost)\n\nfrom tensorflow.config import list_physical_devices  # check if GPU is available","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-22T17:39:58.370896Z","iopub.execute_input":"2022-10-22T17:39:58.371437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training and cross validation\nGPU = list_physical_devices('GPU') != []\nN_ESTIMATORS = 1500\nMAX_DEPTH = 10\nLEARNING_RATE = 0.01\nFOLDS = 5\n\n# data loading\nDEBUG = False\nSAMPLE = 0.2\nSEED = 42","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncol_dtypes = {\n    'game_num': 'int8', 'event_id': 'int8', 'event_time': 'float16',\n    'ball_pos_x': 'float16', 'ball_pos_y': 'float16', 'ball_pos_z': 'float16',\n    'ball_vel_x': 'float16', 'ball_vel_y': 'float16', 'ball_vel_z': 'float16',\n    'p0_pos_x': 'float16', 'p0_pos_y': 'float16', 'p0_pos_z': 'float16',\n    'p0_vel_x': 'float16', 'p0_vel_y': 'float16', 'p0_vel_z': 'float16',\n    'p0_boost': 'float16', 'p1_pos_x': 'float16', 'p1_pos_y': 'float16',\n    'p1_pos_z': 'float16', 'p1_vel_x': 'float16', 'p1_vel_y': 'float16',\n    'p1_vel_z': 'float16', 'p1_boost': 'float16', 'p2_pos_x': 'float16',\n    'p2_pos_y': 'float16', 'p2_pos_z': 'float16', 'p2_vel_x': 'float16',\n    'p2_vel_y': 'float16', 'p2_vel_z': 'float16', 'p2_boost': 'float16',\n    'p3_pos_x': 'float16', 'p3_pos_y': 'float16', 'p3_pos_z': 'float16',\n    'p3_vel_x': 'float16', 'p3_vel_y': 'float16', 'p3_vel_z': 'float16',\n    'p3_boost': 'float16', 'p4_pos_x': 'float16', 'p4_pos_y': 'float16',\n    'p4_pos_z': 'float16', 'p4_vel_x': 'float16', 'p4_vel_y': 'float16',\n    'p4_vel_z': 'float16', 'p4_boost': 'float16', 'p5_pos_x': 'float16',\n    'p5_pos_y': 'float16', 'p5_pos_z': 'float16', 'p5_vel_x': 'float16',\n    'p5_vel_y': 'float16', 'p5_vel_z': 'float16', 'p5_boost': 'float16',\n    'boost0_timer': 'float16', 'boost1_timer': 'float16', 'boost2_timer': 'float16',\n    'boost3_timer': 'float16', 'boost4_timer': 'float16', 'boost5_timer': 'float16',\n    'player_scoring_next': 'O', 'team_scoring_next': 'O', 'team_A_scoring_within_10sec': 'O',\n    'team_B_scoring_within_10sec': 'O'\n}\ncols = list(col_dtypes.keys())\n\npath_to_data = '../input/tabular-playground-series-oct-2022'\ndf = pd.DataFrame({}, columns=cols)\nfor i in range(10):\n    df_tmp = pd.read_csv(f'{path_to_data}/train_{i}.csv', dtype=col_dtypes)\n    if SAMPLE < 1:\n        df_tmp = df_tmp.sample(frac=SAMPLE, random_state=SEED)\n        \n    df = pd.concat([df, df_tmp])\n    del df_tmp\n    gc.collect()\n    if DEBUG:\n        break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def euclidian_norm(x):\n    return np.linalg.norm(x, axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's group the x, y and z variables by player and ball categories\n# this simplifies the code for the euclidian norm calculation\nvel_groups = {\n    f\"{el}_vel\": [f'{el}_vel_x', f'{el}_vel_y', f'{el}_vel_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups = {\n    f\"{el}_pos\": [f'{el}_pos_x', f'{el}_pos_y', f'{el}_pos_z']\n    for el in ['ball'] + [f'p{i}' for i in range(6)]\n}\npos_groups","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# velocity magnitude\nfor col, vec in vel_groups.items():\n    df[col] = euclidian_norm(df[vec])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distance from ball\nfor col, vec in pos_groups.items():\n    df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_drop = [\n    'game_num', 'event_id', 'event_time', 'player_scoring_next', 'team_scoring_next', 'ball_pos_ball_dist'\n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(columns=cols_to_drop)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_p0_pos_x_count = df['p0_pos_x'].isna().sum()\nnull_p0_pos_x_perc = null_p0_pos_x_count / df.shape[0]\nprint(f\"Missing {null_p0_pos_x_count} values ({null_p0_pos_x_perc:.2%})\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(axis=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"has_na = {}\nfor col in df.columns:\n    has_na[col] = df[col].isnull().values.any()\n\nprint(\"Columns that contain null values:\")\nfor col in has_na:\n    if has_na[col]:\n        print(col)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# used to encode the binary classes\nle_a = LabelEncoder()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_a = XGBClassifier(\n    n_estimators=N_ESTIMATORS,\n    max_depth=MAX_DEPTH,\n    learning_rate=LEARNING_RATE,\n    objective='binary:logistic',\n    tree_method='gpu_hist' if GPU else 'hist'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_a = cross_validate(\n    model_a, \n    X=df.drop(columns=['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']).values,\n    y=le_a.fit_transform(df['team_A_scoring_within_10sec'].values),\n    scoring=\"neg_log_loss\",\n    cv=FOLDS,\n    verbose=2,\n    return_estimator=True\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# used to encode the binary classes\nle_b = LabelEncoder()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_b = XGBClassifier(\n    n_estimators=N_ESTIMATORS,\n    max_depth=MAX_DEPTH,\n    learning_rate=LEARNING_RATE,\n    objective='binary:logistic',\n    tree_method='gpu_hist' if GPU else 'hist'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_b = cross_validate(\n    model_b, \n    X=df.drop(columns=['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']).values,\n    y=le_b.fit_transform(df['team_B_scoring_within_10sec'].values),\n    scoring=\"neg_log_loss\",\n    cv=FOLDS,\n    verbose=2,\n    return_estimator=True\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cv_a = pd.DataFrame(\n    {\n        \"model\": \"Model A\",\n        \"fold\": list(range(FOLDS)),\n        \"test_log_loss\": - cv_a[\"test_score\"]\n    }\n)\ndf_cv_b = pd.DataFrame(\n    {\n        \"model\": \"Model B\",\n        \"fold\": list(range(FOLDS)),\n        \"test_log_loss\": - cv_b[\"test_score\"]\n    }\n)\ndf_cv = pd.concat([df_cv_a, df_cv_b])\n\ndel df_cv_a\ndel df_cv_b\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.bar(\n    df_cv, x='fold', y='test_log_loss', color='model', \n    barmode='group', title='Cross Validation Log Loss'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\ndf_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(df):\n    # velocity magnitude\n    for col, vec in vel_groups.items():\n        df[col] = euclidian_norm(df[vec])\n    \n    # ball distance\n    for col, vec in pos_groups.items():\n        df[col + \"_ball_dist\"] = euclidian_norm(df[vec].values - df[pos_groups[\"ball_pos\"]].values)\n    \n    df = df.drop(columns=['ball_pos_ball_dist'])\n    \n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = preprocess(df_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# take the mean of the predictions made by the k models gotten out of the k-fold cross validation\npred_a = np.zeros(df_test.shape[0])\nfor estimator in cv_a['estimator']:\n    pred_a += estimator.predict_proba(df_test.drop(columns=['id']).values)[:, 1]\n\npred_a /= FOLDS","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# take the mean of the predictions made by the k models gotten out of the k-fold cross validation\npred_b = np.zeros(df_test.shape[0])\nfor estimator in cv_b['estimator']:\n    pred_b += estimator.predict_proba(df_test.drop(columns=['id']).values)[:, 1]\n\npred_b /= FOLDS","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        \"id\": df_test['id'],\n        \"team_A_scoring_within_10sec\": pred_a,\n        \"team_B_scoring_within_10sec\": pred_b\n    }\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}