{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is my first notebook for a Kaggle competition. In this I pretend to practice the TF skills gained during the period of training in the Google Developer Machine Learning Bootcamp.\n\nNew things learned from other notebooks:\n* Feather format to retrieve data\n\nDataset used: https://www.kaggle.com/datasets/eavelardev/rocket-league","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport pandas as pd\nimport tensorflow as tf\nimport numpy as np\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:32.634566Z","iopub.execute_input":"2022-10-31T22:25:32.635001Z","iopub.status.idle":"2022-10-31T22:25:35.389855Z","shell.execute_reply.started":"2022-10-31T22:25:32.634916Z","shell.execute_reply":"2022-10-31T22:25:35.388531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.391684Z","iopub.execute_input":"2022-10-31T22:25:35.392365Z","iopub.status.idle":"2022-10-31T22:25:35.398444Z","shell.execute_reply.started":"2022-10-31T22:25:35.392324Z","shell.execute_reply":"2022-10-31T22:25:35.397054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data source","metadata":{}},{"cell_type":"code","source":"data_path = '/kaggle/input/tabular-playground-series-oct-2022/'\nfeather_path = '/kaggle/input/rocket-league/'","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.400485Z","iopub.execute_input":"2022-10-31T22:25:35.401055Z","iopub.status.idle":"2022-10-31T22:25:35.414993Z","shell.execute_reply.started":"2022-10-31T22:25:35.400994Z","shell.execute_reply":"2022-10-31T22:25:35.413954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dtypes_file = os.path.join(data_path, 'train_dtypes.csv')\ntest_dtypes_file = os.path.join(data_path, 'test_dtypes.csv')\nsample_submission_file = os.path.join(data_path, 'sample_submission.csv')\ntrain_feather_file = os.path.join(feather_path, 'train.feather')\ntest_feather_file = os.path.join(feather_path, 'test.feather')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.418737Z","iopub.execute_input":"2022-10-31T22:25:35.420043Z","iopub.status.idle":"2022-10-31T22:25:35.435496Z","shell.execute_reply.started":"2022-10-31T22:25:35.419986Z","shell.execute_reply":"2022-10-31T22:25:35.434340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data analysis","metadata":{}},{"cell_type":"code","source":"train_dtypes_df = pd.read_csv(train_dtypes_file)\ntest_dtypes_df = pd.read_csv(test_dtypes_file)\ncols_dtypes = {k: v for (k, v) in zip(train_dtypes_df.column, train_dtypes_df.dtype)}","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.437022Z","iopub.execute_input":"2022-10-31T22:25:35.438388Z","iopub.status.idle":"2022-10-31T22:25:35.459419Z","shell.execute_reply.started":"2022-10-31T22:25:35.438345Z","shell.execute_reply":"2022-10-31T22:25:35.457842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# columns only in trainig\n[c for c in train_dtypes_df['column'] if c not in test_dtypes_df['column'].values]","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.461267Z","iopub.execute_input":"2022-10-31T22:25:35.461920Z","iopub.status.idle":"2022-10-31T22:25:35.475074Z","shell.execute_reply.started":"2022-10-31T22:25:35.461878Z","shell.execute_reply":"2022-10-31T22:25:35.473762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature selection","metadata":{}},{"cell_type":"code","source":"targets = ['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']\nuseless_cols = ['game_num', 'event_time', 'player_scoring_next', 'team_scoring_next']\nuse_cols = [c for c in train_dtypes_df['column'] if c not in useless_cols]\nfeatures = test_dtypes_df['column'][1:].values # drop id","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.476873Z","iopub.execute_input":"2022-10-31T22:25:35.478093Z","iopub.status.idle":"2022-10-31T22:25:35.487460Z","shell.execute_reply.started":"2022-10-31T22:25:35.478037Z","shell.execute_reply":"2022-10-31T22:25:35.486189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    train_dfs = []\n    for i in range(10):\n        train_file = os.path.join(data_path, f'train_{i}.csv')\n        train_dfs.append(pd.read_csv(train_file, usecols=use_cols, dtype=cols_dtypes))\n\n    pd.concat(train_dfs).reset_index(drop=True).to_feather('train.feather')\n    del train_dfs\n    \n    test_file = os.path.join(data_path, 'test.csv')\n    df = pd.read_csv(test_file, usecols=features, dtype=cols_dtypes)\n    df.to_feather('test.feather')\n    del df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.489224Z","iopub.execute_input":"2022-10-31T22:25:35.489841Z","iopub.status.idle":"2022-10-31T22:25:35.502537Z","shell.execute_reply.started":"2022-10-31T22:25:35.489789Z","shell.execute_reply":"2022-10-31T22:25:35.501284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_feather(train_feather_file)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:35.504021Z","iopub.execute_input":"2022-10-31T22:25:35.504443Z","iopub.status.idle":"2022-10-31T22:25:56.042802Z","shell.execute_reply.started":"2022-10-31T22:25:35.504402Z","shell.execute_reply":"2022-10-31T22:25:56.041084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering and Normalization","metadata":{}},{"cell_type":"code","source":"# get team goals\nbal_pos_cols = ['ball_pos_x', 'ball_pos_y', 'ball_pos_z']\ngoal_pos_cols = ['goal_pos_x', 'goal_pos_y', 'goal_pos_z']\nlast_event_samples = train_df['event_id'].diff(periods=-1)\n\nteam_A_goal = train_df[(last_event_samples != 0) & (train_df['team_A_scoring_within_10sec'] == 1)][bal_pos_cols].mean()\nteam_B_goal = train_df[(last_event_samples != 0) & (train_df['team_B_scoring_within_10sec'] == 1)][bal_pos_cols].mean()\ngoals = pd.DataFrame({'team_A_goal': team_A_goal.values, \n                      'team_B_goal': team_B_goal.values}, \n                      index=goal_pos_cols).transpose().abs().mean().round(1)\nteam_A_goal = goals.copy()\nteam_B_goal = goals.copy()\nteam_B_goal['goal_pos_y'] *= -1\n\ntrain_df.drop('event_id', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:56.046025Z","iopub.execute_input":"2022-10-31T22:25:56.046543Z","iopub.status.idle":"2022-10-31T22:25:59.251114Z","shell.execute_reply.started":"2022-10-31T22:25:56.046463Z","shell.execute_reply":"2022-10-31T22:25:59.249952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame({'team_A_goal': team_A_goal.values, \n              'team_B_goal': team_B_goal.values}, \n              index=goal_pos_cols)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:59.252794Z","iopub.execute_input":"2022-10-31T22:25:59.253383Z","iopub.status.idle":"2022-10-31T22:25:59.271901Z","shell.execute_reply.started":"2022-10-31T22:25:59.253332Z","shell.execute_reply":"2022-10-31T22:25:59.270216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets_df = pd.DataFrame()\ntargets_df['team_A_scoring_within_10sec'] = train_df.pop('team_A_scoring_within_10sec')\ntargets_df['team_B_scoring_within_10sec'] = train_df.pop('team_B_scoring_within_10sec')\ntargets_df['no_team_scored'] = (\n    targets_df['team_A_scoring_within_10sec'] == targets_df['team_B_scoring_within_10sec']\n).astype('int8')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:25:59.273331Z","iopub.execute_input":"2022-10-31T22:25:59.273976Z","iopub.status.idle":"2022-10-31T22:26:01.740692Z","shell.execute_reply.started":"2022-10-31T22:25:59.273938Z","shell.execute_reply":"2022-10-31T22:26:01.739269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_normalized_df(df, min_data, max_data):\n    df.fillna(0, inplace=True)\n    cols_to_drop = []\n    \n    # speed\n    for i in range(6):\n        df[f'p{i}_speed'] = np.sqrt(\n           df[f'p{i}_vel_x']**2+\n           df[f'p{i}_vel_y']**2+\n           df[f'p{i}_vel_z']**2)\n        cols_to_drop.extend([f'p{i}_vel_x', f'p{i}_vel_y', f'p{i}_vel_z'])\n        \n    df['ball_speed'] = np.sqrt(\n       df['ball_vel_x']**2+\n       df['ball_vel_y']**2+\n       df['ball_vel_z']**2) \n    cols_to_drop.extend(['ball_vel_x', 'ball_vel_y', 'ball_vel_z'])\n    \n    # distances\n    for i in range(6):\n        df[f'p{i}_dist_ball'] = np.sqrt(\n           (df[f'p{i}_pos_x']-df['ball_pos_x'])**2+\n           (df[f'p{i}_pos_y']-df['ball_pos_y'])**2+\n           (df[f'p{i}_pos_z']-df['ball_pos_z'])**2)\n        cols_to_drop.extend([f'p{i}_pos_x', f'p{i}_pos_y', f'p{i}_pos_z'])\n\n    for i in range(3):\n        df[f'p{i}_dist_goal'] = np.sqrt(\n           (df[f'p{i}_pos_x']-team_A_goal['goal_pos_x'])**2+\n           (df[f'p{i}_pos_y']-team_A_goal['goal_pos_y'])**2+\n           (df[f'p{i}_pos_z']-team_A_goal['goal_pos_z'])**2)\n    \n    for i in range(3, 6):\n        df[f'p{i}_dist_goal'] = np.sqrt(\n           (df[f'p{i}_pos_x']-team_B_goal['goal_pos_x'])**2+\n           (df[f'p{i}_pos_y']-team_B_goal['goal_pos_y'])**2+\n           (df[f'p{i}_pos_z']-team_B_goal['goal_pos_z'])**2)\n    \n    df['ball_dist_goal_A'] = np.sqrt(\n           (df['ball_pos_x']-team_A_goal['goal_pos_x'])**2+\n           (df['ball_pos_y']-team_A_goal['goal_pos_y'])**2+\n           (df['ball_pos_z']-team_A_goal['goal_pos_z'])**2)\n    \n    df['ball_dist_goal_B'] = np.sqrt(\n           (df['ball_pos_x']-team_B_goal['goal_pos_x'])**2+\n           (df['ball_pos_y']-team_B_goal['goal_pos_y'])**2+\n           (df['ball_pos_z']-team_B_goal['goal_pos_z'])**2) \n    \n    cols_to_drop.extend(['ball_pos_x', 'ball_pos_y', 'ball_pos_z'])\n    \n    df.drop(cols_to_drop, axis=1, inplace=True)\n    \n    if min_data is None:\n        min_data = df.min()\n        max_data = df.max()\n        \n        components = ['pos', 'vel', 'speed', 'dist']\n\n        # same max min by component\n        for comp in components:\n            min_data[min_data.filter(like=comp).index] = min_data.filter(like=comp).min()\n            max_data[max_data.filter(like=comp).index] = max_data.filter(like=comp).max()\n\n    df = (df - min_data)/(max_data - min_data)\n    \n    return df, min_data, max_data","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:01.747004Z","iopub.execute_input":"2022-10-31T22:26:01.747460Z","iopub.status.idle":"2022-10-31T22:26:01.766881Z","shell.execute_reply.started":"2022-10-31T22:26:01.747418Z","shell.execute_reply":"2022-10-31T22:26:01.765090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_data = None\nmax_data = None\ntrain_df, min_data, max_data = get_normalized_df(train_df, min_data, max_data)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:01.768714Z","iopub.execute_input":"2022-10-31T22:26:01.769255Z","iopub.status.idle":"2022-10-31T22:26:39.899775Z","shell.execute_reply.started":"2022-10-31T22:26:01.769198Z","shell.execute_reply":"2022-10-31T22:26:39.898188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X, val_X, train_y, val_y = train_test_split(train_df, targets_df, test_size=0.007, random_state=0)\ndel (train_df, targets_df)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:39.901257Z","iopub.execute_input":"2022-10-31T22:26:39.901627Z","iopub.status.idle":"2022-10-31T22:26:56.422724Z","shell.execute_reply.started":"2022-10-31T22:26:39.901594Z","shell.execute_reply":"2022-10-31T22:26:56.421399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.Sequential([\n    layers.BatchNormalization(input_shape = [len(train_X.columns)]),\n    layers.Dense(64, activation='relu'),\n    \n    layers.BatchNormalization(),\n    layers.Dropout(0.05),\n    layers.Dense(128, activation='relu'),\n    \n    layers.BatchNormalization(),\n    layers.Dropout(0.1),\n    layers.Dense(128, activation='relu'),\n \n    layers.BatchNormalization(),\n    layers.Dropout(0.15),\n    layers.Dense(64, activation='relu'),    \n    \n    layers.BatchNormalization(),\n    layers.Dropout(0.2),\n    layers.Dense(3)\n])","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:56.424317Z","iopub.execute_input":"2022-10-31T22:26:56.425044Z","iopub.status.idle":"2022-10-31T22:26:56.690006Z","shell.execute_reply.started":"2022-10-31T22:26:56.425001Z","shell.execute_reply":"2022-10-31T22:26:56.688455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',\n             loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True),\n             metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:56.691507Z","iopub.execute_input":"2022-10-31T22:26:56.691869Z","iopub.status.idle":"2022-10-31T22:26:56.709979Z","shell.execute_reply.started":"2022-10-31T22:26:56.691836Z","shell.execute_reply":"2022-10-31T22:26:56.708923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    keras.callbacks.EarlyStopping(\n        monitor=\"val_loss\",\n        patience=8,\n        restore_best_weights=True)\n]","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:56.711894Z","iopub.execute_input":"2022-10-31T22:26:56.712566Z","iopub.status.idle":"2022-10-31T22:26:56.716731Z","shell.execute_reply.started":"2022-10-31T22:26:56.712513Z","shell.execute_reply":"2022-10-31T22:26:56.715865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = model.fit(train_X, train_y, \n              epochs=150, \n              steps_per_epoch=5000, \n              batch_size=32,\n              validation_data=(val_X, val_y),\n              callbacks=[callbacks])\n\ndel (train_X, val_X, train_y, val_y)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:56.718274Z","iopub.execute_input":"2022-10-31T22:26:56.718628Z","iopub.status.idle":"2022-10-31T22:26:56.936394Z","shell.execute_reply.started":"2022-10-31T22:26:56.718594Z","shell.execute_reply":"2022-10-31T22:26:56.934819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_feather(test_feather_file)\ntest_df.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:56.938368Z","iopub.execute_input":"2022-10-31T22:26:56.938798Z","iopub.status.idle":"2022-10-31T22:26:58.390402Z","shell.execute_reply.started":"2022-10-31T22:26:56.938760Z","shell.execute_reply":"2022-10-31T22:26:58.389191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df, _, _ = get_normalized_df(test_df, min_data, max_data)\ntest_df.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:58.391691Z","iopub.execute_input":"2022-10-31T22:26:58.392013Z","iopub.status.idle":"2022-10-31T22:26:58.963465Z","shell.execute_reply.started":"2022-10-31T22:26:58.391984Z","shell.execute_reply":"2022-10-31T22:26:58.962340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_df)\nscore = tf.nn.softmax(preds)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:26:58.967118Z","iopub.execute_input":"2022-10-31T22:26:58.967502Z","iopub.status.idle":"2022-10-31T22:27:42.450502Z","shell.execute_reply.started":"2022-10-31T22:26:58.967467Z","shell.execute_reply":"2022-10-31T22:27:42.449050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(sample_submission_file)\nss['team_A_scoring_within_10sec'] = score[:,0]\nss['team_B_scoring_within_10sec'] = score[:,1]\nss.to_csv('Submission.csv', index=False)\nss.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T22:27:42.452489Z","iopub.execute_input":"2022-10-31T22:27:42.453388Z","iopub.status.idle":"2022-10-31T22:27:44.533666Z","shell.execute_reply.started":"2022-10-31T22:27:42.453335Z","shell.execute_reply":"2022-10-31T22:27:44.532316Z"},"trusted":true},"execution_count":null,"outputs":[]}]}