{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-28T14:53:29.149054Z","iopub.execute_input":"2022-10-28T14:53:29.149604Z","iopub.status.idle":"2022-10-28T14:53:29.182448Z","shell.execute_reply.started":"2022-10-28T14:53:29.149475Z","shell.execute_reply":"2022-10-28T14:53:29.181594Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport seaborn as sns\nsns.set(style='darkgrid', font_scale=1.6)\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\npd.options.display.max_columns = None\nfrom sklearn.model_selection import train_test_split\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:53:29.209748Z","iopub.execute_input":"2022-10-28T14:53:29.210419Z","iopub.status.idle":"2022-10-28T14:53:37.407857Z","shell.execute_reply.started":"2022-10-28T14:53:29.210382Z","shell.execute_reply":"2022-10-28T14:53:37.406574Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Setting paths for original dataset and to save feather files\n\ndata_path = '/kaggle/input/tabular-playground-series-oct-2022/'\nfeather_path = './'\n\n#Train-test files for original and feather dataset\ntrain_dtypes_file = os.path.join(data_path, 'train_dtypes.csv')\ntest_dtypes_file = os.path.join(data_path, 'test_dtypes.csv')\nsample_submission_file = os.path.join(data_path, 'sample_submission.csv')\ntrain_feather_file = os.path.join(feather_path, 'train.feather')\ntest_feather_file = os.path.join(feather_path, 'test.feather')\n\ntrain_dtypes_df = pd.read_csv(train_dtypes_file)\ntest_dtypes_df = pd.read_csv(test_dtypes_file)\ncols_dtypes = {k: v for (k, v) in zip(train_dtypes_df.column, train_dtypes_df.dtype)}\ncols_test_dtypes = {k: v for (k, v) in zip(test_dtypes_df.column, test_dtypes_df.dtype)}","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:53:37.446356Z","iopub.execute_input":"2022-10-28T14:53:37.447409Z","iopub.status.idle":"2022-10-28T14:53:37.481744Z","shell.execute_reply.started":"2022-10-28T14:53:37.447197Z","shell.execute_reply":"2022-10-28T14:53:37.480753Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Introduction\n\nThe purpose of this notebook is to generate a compressed **Rocket League Dataset**. We will take each train file and, after applying feature engineering, they will be saved in feather extension to save memory while loading these data. \n\n## Disclaimer\nIdeas for compressing data using .feather and feature engineering were taken from [TPS-OCT-2022 - Simple TF](https://www.kaggle.com/code/eavelardev/tps-oct-2022-simple-tf/notebook) by Eduardo Avelar.","metadata":{}},{"cell_type":"markdown","source":"# 2. Procedure","metadata":{}},{"cell_type":"code","source":"useless_cols = ['team_scoring_next', 'player_scoring_next', 'event_time', 'event_id']\nuse_cols = [c for c in train_dtypes_df['column'] if c not in useless_cols]","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:53:56.897839Z","iopub.execute_input":"2022-10-28T14:53:56.898255Z","iopub.status.idle":"2022-10-28T14:53:56.904341Z","shell.execute_reply.started":"2022-10-28T14:53:56.898222Z","shell.execute_reply":"2022-10-28T14:53:56.903147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to add number of active players for both teams.**","metadata":{}},{"cell_type":"code","source":"def demolitions(df):\n    df['active_players_A'] = 3-(df['p0_pos_x'].isna()).astype(int)-(df['p1_pos_x'].isna()).astype(int)-(df['p2_pos_x'].isna()).astype(int)\n    df['active_players_B'] = 3-(df['p3_pos_x'].isna()).astype(int)-(df['p4_pos_x'].isna()).astype(int)-(df['p5_pos_x'].isna()).astype(int)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:53:58.984709Z","iopub.execute_input":"2022-10-28T14:53:58.985135Z","iopub.status.idle":"2022-10-28T14:53:58.992518Z","shell.execute_reply.started":"2022-10-28T14:53:58.985098Z","shell.execute_reply":"2022-10-28T14:53:58.991621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to add new variables:**\n\n- Distance between Ball and both goals\n- Ball speed\n- Players speed\n- Distance between team A players and team B goal\n- Distance between team B players and team A goal\n- Players to ball distance\n- Missing values are filled with 0. This was the easiest solution. More complex solutions did not show better results.\n\nNotes\n   - All distances are calculated using euclidean distance\n   - Goal coordinates are estimates calculated in [TPS-OCT-2022 - Simple TF](https://www.kaggle.com/code/eavelardev/tps-oct-2022-simple-tf/notebook)","metadata":{}},{"cell_type":"code","source":"#Computes distances and speeds.\ndef compute_df(df):\n    # Estimates\n    goalA_coord = (0,-103.5, 6)\n    goalB_coord = (0,103.5, 6)\n    df.fillna(0, inplace=True)\n    \n    # Euclidean distance. Ball to goals distances\n    df['ball_dist_to_goalA'] = np.sqrt((df['ball_pos_x']-goalA_coord[0])**2 + (df['ball_pos_y']-goalA_coord[1])**2 + (df['ball_pos_z']-goalA_coord[2])**2)\n    df['ball_dist_to_goalB'] = np.sqrt((df['ball_pos_x']-goalB_coord[0])**2 + (df['ball_pos_y']-goalB_coord[1])**2 + (df['ball_pos_z']-goalB_coord[2])**2)\n    \n    #Ball speed\n    df['ball_speed'] = np.sqrt(df['ball_vel_x']**2 + df['ball_vel_y']**2 + df['ball_vel_z']**2) \n    \n    #Players speeds\n    for i in range(6):\n        df[f'p{i}_speed'] = np.sqrt(df[f'p{i}_vel_x']**2 + df[f'p{i}_vel_y']**2 + df[f'p{i}_vel_z']**2)\n        \n        \n    #Team A distance to Goal B\n    for i in range(3):\n        df[f'p{i}_dist_to_goal'] = np.sqrt((df[f'p{i}_pos_x'] - goalB_coord[0])**2 +\n                                           (df[f'p{i}_pos_y'] - goalB_coord[1])**2 +\n                                           (df[f'p{i}_pos_z'] - goalB_coord[2])**2)\n        \n        \n    #Team B distance to Goal A\n    for i in range(3, 6):\n        df[f'p{i}_dist_to_goal'] = np.sqrt((df[f'p{i}_pos_x'] - goalA_coord[0])**2 +\n                                           (df[f'p{i}_pos_y'] - goalA_coord[1])**2 +\n                                           (df[f'p{i}_pos_z'] - goalA_coord[2])**2)\n        \n        \n    #Player to Ball distance\n    \n    for i in range(6):\n        df[f'p{i}_dist_to_ball'] = np.sqrt((df[f'p{i}_pos_x']-df['ball_pos_x'])**2 + (df[f'p{i}_pos_y']-df['ball_pos_y'])**2 + (df[f'p{i}_pos_z']-df['ball_pos_z'])**2)\n        \n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:54:01.261917Z","iopub.execute_input":"2022-10-28T14:54:01.262340Z","iopub.status.idle":"2022-10-28T14:54:01.279195Z","shell.execute_reply.started":"2022-10-28T14:54:01.262305Z","shell.execute_reply":"2022-10-28T14:54:01.277952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Iteration over each file**","metadata":{}},{"cell_type":"code","source":"for i in range(10):\n    train_file = os.path.join(data_path, f'train_{i}.csv')\n    train = pd.read_csv(train_file, usecols=use_cols, dtype=cols_dtypes)\n    \n    train = demolitions(train)\n    train = compute_df(train)\n    \n    train.to_feather(f'train_{i}.feather')\n    del train\n\n\ntest_file = os.path.join(data_path, 'test.csv')\ndf = pd.read_csv(test_file, dtype=cols_test_dtypes)\ndf.to_feather('test.feather')\ndel df","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:54:13.827197Z","iopub.execute_input":"2022-10-28T14:54:13.827646Z","iopub.status.idle":"2022-10-28T15:00:29.562262Z","shell.execute_reply.started":"2022-10-28T14:54:13.827609Z","shell.execute_reply":"2022-10-28T15:00:29.560685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Dataset\n\nWith the output feather files a new dataset is generated for further development.","metadata":{}}]}