{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Inspiration\nI drew some inspiration from the following notebooks:\n\n - [@alexryzhkov's How to kill all your efforts?](https://www.kaggle.com/code/alexryzhkov/how-to-kill-all-your-efforts)\n - [@cristiansanabria's Data shuffled and split in 10 feather files](https://www.kaggle.com/code/cristiansanabria/data-shuffled-and-split-in-10-feather-files)\n - [@shoooono's Oct2022_Feature_Engineering](https://www.kaggle.com/code/shoooono/oct2022-feature-engineering) ","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\nimport itertools\n\npd.set_option('display.max_columns', 500)\npd.set_option('display.max_rows', 100)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:50:34.589609Z","iopub.execute_input":"2022-10-31T01:50:34.590068Z","iopub.status.idle":"2022-10-31T01:50:34.615984Z","shell.execute_reply.started":"2022-10-31T01:50:34.589983Z","shell.execute_reply":"2022-10-31T01:50:34.614948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"player_features = [\n    ['p0_pos_x', 'p0_pos_y', 'p0_pos_z','p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost'],\n    ['p1_pos_x', 'p1_pos_y', 'p1_pos_z','p1_vel_x', 'p1_vel_y', 'p1_vel_z', 'p1_boost'],\n    ['p2_pos_x', 'p2_pos_y', 'p2_pos_z','p2_vel_x', 'p2_vel_y', 'p2_vel_z', 'p2_boost'],\n    ['p3_pos_x', 'p3_pos_y', 'p3_pos_z','p3_vel_x', 'p3_vel_y', 'p3_vel_z', 'p3_boost'],\n    ['p4_pos_x', 'p4_pos_y', 'p4_pos_z','p4_vel_x', 'p4_vel_y', 'p4_vel_z', 'p4_boost'],\n    ['p5_pos_x', 'p5_pos_y', 'p5_pos_z','p5_vel_x', 'p5_vel_y', 'p5_vel_z', 'p5_boost'],\n]\n\nball_features = [\n    'ball_pos_x', 'ball_pos_y','ball_pos_z','ball_vel_x', 'ball_vel_y', 'ball_vel_z'\n]\n\nboost_timer_features = [\n    'boost0_timer', 'boost1_timer', 'boost2_timer', \n    'boost3_timer','boost4_timer', 'boost5_timer',\n]\n\nNUMERICAL_FEATURES = ball_features + boost_timer_features + list(itertools.chain(*player_features))\nMETA_FEATURES = [\"game_num\", \"event_id\"]\nLABELS = [\"team_A_scoring_within_10sec\", \"team_B_scoring_within_10sec\"]\n\nALL_COLS = NUMERICAL_FEATURES + META_FEATURES + LABELS","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:50:34.623678Z","iopub.execute_input":"2022-10-31T01:50:34.624069Z","iopub.status.idle":"2022-10-31T01:50:34.640198Z","shell.execute_reply.started":"2022-10-31T01:50:34.624034Z","shell.execute_reply":"2022-10-31T01:50:34.638952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List input files\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-31T01:50:34.648244Z","iopub.execute_input":"2022-10-31T01:50:34.648588Z","iopub.status.idle":"2022-10-31T01:50:34.655876Z","shell.execute_reply.started":"2022-10-31T01:50:34.648558Z","shell.execute_reply":"2022-10-31T01:50:34.654646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compress files to feather format\nFEATHER_INPUT_DIR = \"/kaggle/working/feather-data/\"\n\ndef compress_data_to_feather(indexes: list = None, overwrite=False):\n    if os.path.exists(FEATHER_INPUT_DIR):\n        print(\"Already created the feather-data directory\")\n    else:\n        os.makedirs(FEATHER_INPUT_DIR)\n        print(\"Created feather-data directory\")\n        \n    if indexes is None: # Operate on all the datasets\n        indexes = list(range(10))\n    \n    train_dtypes = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv\")\n    train_dtypes: dict = dict(train_dtypes.to_records(index=False))\n            \n    for indx in indexes:    \n        file_name = f\"train_{indx}\"\n        \n        if not overwrite and os.path.exists(FEATHER_INPUT_DIR + f\"{file_name}.ftr\"):\n            print(f\"File already compressed and overwrite=False, skipping {file_name}.csv\")\n            continue\n        \n        tr_df = pd.read_csv(f\"/kaggle/input/tabular-playground-series-oct-2022/{file_name}.csv\", dtype=train_dtypes, usecols=ALL_COLS)\n        \n        mem_usage = round(tr_df.memory_usage().sum()/(1024**2), 3)\n        print(f\"Reading Data Memory Usage: {mem_usage}MB for {file_name}\")\n        \n        tr_df.to_feather(FEATHER_INPUT_DIR + f\"{file_name}.ftr\")        \n        del tr_df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:50:34.661561Z","iopub.execute_input":"2022-10-31T01:50:34.662058Z","iopub.status.idle":"2022-10-31T01:50:34.671871Z","shell.execute_reply.started":"2022-10-31T01:50:34.662022Z","shell.execute_reply":"2022-10-31T01:50:34.670670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compress_data_to_feather()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:50:34.674092Z","iopub.execute_input":"2022-10-31T01:50:34.674457Z","iopub.status.idle":"2022-10-31T01:55:31.303133Z","shell.execute_reply.started":"2022-10-31T01:50:34.674426Z","shell.execute_reply":"2022-10-31T01:55:31.301779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's work with the first dataset\ntr_df = pd.read_feather(FEATHER_INPUT_DIR + \"/train_5.ftr\")\ntr_df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:31.304956Z","iopub.execute_input":"2022-10-31T01:55:31.305335Z","iopub.status.idle":"2022-10-31T01:55:31.844182Z","shell.execute_reply.started":"2022-10-31T01:55:31.305303Z","shell.execute_reply":"2022-10-31T01:55:31.843030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE ENGINEERING\n  \n  -  Distance of Ball to the Net/Goals (`dist_ball_a_net`, `dist_ball_b_net`)\n  -  Distance of Ball to the Players\n  -  Distance of Players to the Net/Goals (`dist_A_net`, `dist_B_net`)\n \n\nThese features are rather interesting. For example, we can tell which players didn't really interact with the ball (distance to the ball was never <= 5.0).\n\n  ","metadata":{}},{"cell_type":"code","source":"A_goal_coords = (0, -104, 0)\nB_goal_coords = (0, 104, 0)\n\nPOS_X_MIN = -80\nPOS_Y_MIN = -104\nPOS_Z_MIN = 0\n\nPOS_X_MAX = 80\nPOS_Y_MAX = 104\nPOS_Z_MAX = np.inf\n\nprint(\"\\nMax:\\n\", tr_df[[i for i in tr_df.columns if \"ball\" in i]].max(),\n      \"\\n\\nMin:\\n\",tr_df[[i for i in tr_df.columns if \"ball\" in i]].min(),\n     )","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:31.845743Z","iopub.execute_input":"2022-10-31T01:55:31.846392Z","iopub.status.idle":"2022-10-31T01:55:31.933833Z","shell.execute_reply.started":"2022-10-31T01:55:31.846357Z","shell.execute_reply":"2022-10-31T01:55:31.932763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_dist(p1_x, p1_y, p1_z, p2_x, p2_y, p2_z) -> float:\n    \"\"\" Euclidean distance of 3d vector \"\"\"\n    return np.sqrt(np.square(p2_x - p1_x) + np.square(p2_y - p1_y) + np.square(p2_z - p1_z))","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:31.935953Z","iopub.execute_input":"2022-10-31T01:55:31.936282Z","iopub.status.idle":"2022-10-31T01:55:31.942085Z","shell.execute_reply.started":"2022-10-31T01:55:31.936253Z","shell.execute_reply":"2022-10-31T01:55:31.940911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_df[\"dist_ball_A_goal\"] = calc_dist(\n        tr_df[\"ball_pos_x\"], \n        tr_df[\"ball_pos_y\"], \n        tr_df[\"ball_pos_z\"], \n        A_goal_coords[0], \n        A_goal_coords[1], \n        A_goal_coords[2]\n    )\n\ntr_df[\"dist_ball_B_goal\"] = calc_dist(\n        tr_df[\"ball_pos_x\"], \n        tr_df[\"ball_pos_y\"], \n        tr_df[\"ball_pos_z\"], \n        B_goal_coords[0], \n        B_goal_coords[1], \n        B_goal_coords[2]\n    )\n\ntr_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:31.943321Z","iopub.execute_input":"2022-10-31T01:55:31.943619Z","iopub.status.idle":"2022-10-31T01:55:32.422251Z","shell.execute_reply.started":"2022-10-31T01:55:31.943591Z","shell.execute_reply":"2022-10-31T01:55:32.421076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"players = [\"p0\", \"p1\", \"p2\", \"p3\", \"p4\", \"p5\"]\nfor p in players:\n    tr_df[f\"{p}_dist_to_A_net\"] = calc_dist(\n        tr_df[f\"{p}_pos_x\"], \n        tr_df[f\"{p}_pos_y\"],\n        tr_df[f\"{p}_pos_z\"], \n        A_goal_coords[0], \n        A_goal_coords[1], \n        A_goal_coords[2]\n    )\n    \n    tr_df[f\"{p}_dist_to_B_net\"] = calc_dist(\n        tr_df[f\"{p}_pos_x\"], \n        tr_df[f\"{p}_pos_y\"], \n        tr_df[f\"{p}_pos_z\"], \n        B_goal_coords[0], \n        B_goal_coords[1], \n        B_goal_coords[2]\n    )\n\ntr_df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:32.423850Z","iopub.execute_input":"2022-10-31T01:55:32.424589Z","iopub.status.idle":"2022-10-31T01:55:33.165607Z","shell.execute_reply.started":"2022-10-31T01:55:32.424546Z","shell.execute_reply":"2022-10-31T01:55:33.164438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize data\n# Min-Max Normalization\n\npos_x_cols = [col for col in tr_df.columns if \"pos_x\" in col]\npos_y_cols = [col for col in tr_df.columns if \"pos_y\" in col]\n\ntr_df[pos_x_cols] = (tr_df[pos_x_cols]  - POS_X_MIN) / (POS_X_MAX - POS_X_MIN)\ntr_df[pos_y_cols] = (tr_df[pos_y_cols]  - POS_Y_MIN) / (POS_Y_MAX - POS_Y_MIN)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:33.166956Z","iopub.execute_input":"2022-10-31T01:55:33.167403Z","iopub.status.idle":"2022-10-31T01:55:33.323361Z","shell.execute_reply.started":"2022-10-31T01:55:33.167372Z","shell.execute_reply":"2022-10-31T01:55:33.322204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NaN features\nplayers = [\"p0\", \"p1\", \"p2\", \"p3\", \"p4\", \"p5\"]\nfor i, player in enumerate(players):\n    player_columns = player_features[i]\n    tr_df[f\"{player}_na\"] = tr_df[player_columns].isnull().any(axis=1).astype(\"float16\")\n    \ntr_df.sample(15)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:33.324836Z","iopub.execute_input":"2022-10-31T01:55:33.325170Z","iopub.status.idle":"2022-10-31T01:55:34.022003Z","shell.execute_reply.started":"2022-10-31T01:55:33.325139Z","shell.execute_reply":"2022-10-31T01:55:34.020855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Note the RuntimeWarning gets raised because of the dtype we're using\nfill = tr_df.select_dtypes(include='number').dropna().mean().to_dict()\n\n# Boost feature still has NaN\nfor k,v in fill.items():\n    if np.isnan(v):\n        fill[k] = 0\n\ntr_df.fillna(fill, inplace=True)\n\ntr_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:34.023705Z","iopub.execute_input":"2022-10-31T01:55:34.024196Z","iopub.status.idle":"2022-10-31T01:55:36.796778Z","shell.execute_reply.started":"2022-10-31T01:55:34.024152Z","shell.execute_reply":"2022-10-31T01:55:36.795757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_df.isna().any().any()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:36.799919Z","iopub.execute_input":"2022-10-31T01:55:36.800225Z","iopub.status.idle":"2022-10-31T01:55:37.027918Z","shell.execute_reply.started":"2022-10-31T01:55:36.800196Z","shell.execute_reply":"2022-10-31T01:55:37.026815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Model\n","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras ","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:37.029196Z","iopub.execute_input":"2022-10-31T01:55:37.029516Z","iopub.status.idle":"2022-10-31T01:55:42.189739Z","shell.execute_reply.started":"2022-10-31T01:55:37.029488Z","shell.execute_reply":"2022-10-31T01:55:42.188642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:42.191090Z","iopub.execute_input":"2022-10-31T01:55:42.191760Z","iopub.status.idle":"2022-10-31T01:55:42.560167Z","shell.execute_reply.started":"2022-10-31T01:55:42.191725Z","shell.execute_reply":"2022-10-31T01:55:42.559140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_train_data(df, n_splits=10):\n    y = df[LABELS].to_numpy()\n    X = df.drop(columns=LABELS+META_FEATURES).to_numpy()\n    groups = df[\"game_num\"].to_numpy()\n    print(y.shape, X.shape)\n\n    gkf = GroupKFold(n_splits=n_splits)\n    \n    for train_index, test_index in gkf.split(X, y, groups=groups):\n        print(len(train_index), len(test_index))\n        yield (\n            (X[train_index], y[train_index]), \n            (X[test_index], y[test_index])\n        )\n        ","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:42.564392Z","iopub.execute_input":"2022-10-31T01:55:42.564768Z","iopub.status.idle":"2022-10-31T01:55:42.573039Z","shell.execute_reply.started":"2022-10-31T01:55:42.564736Z","shell.execute_reply":"2022-10-31T01:55:42.571791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model(checkpoint_path):\n    # Create a callback that saves the model's weights. For a bit of online learning :)    \n    checkpoint_dir = os.path.dirname(checkpoint_path)\n    \n    model = tf.keras.models.Sequential([\n        tf.keras.layers.Dense(16, activation=\"relu\"),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dense(2, activation=\"softmax\"),\n    ])\n    model.compile(\n        optimizer=\"adam\",\n        loss=keras.losses.CategoricalCrossentropy(),\n        metrics=[keras.metrics.CategoricalAccuracy()]\n    )\n    \n    latest = tf.train.latest_checkpoint(checkpoint_dir)\n    if latest is not None:\n        model.load_weights(latest)\n        \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:42.574910Z","iopub.execute_input":"2022-10-31T01:55:42.575331Z","iopub.status.idle":"2022-10-31T01:55:42.586789Z","shell.execute_reply.started":"2022-10-31T01:55:42.575287Z","shell.execute_reply":"2022-10-31T01:55:42.585983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Grab a quick dataset for testing\n(x_train,y_train), (x_test, y_test) = next(get_test_train_data(tr_df, n_splits=3))","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:42.588109Z","iopub.execute_input":"2022-10-31T01:55:42.588991Z","iopub.status.idle":"2022-10-31T01:55:43.861852Z","shell.execute_reply.started":"2022-10-31T01:55:42.588958Z","shell.execute_reply":"2022-10-31T01:55:43.860420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EpochModelCheckpoint(tf.keras.callbacks.ModelCheckpoint):\n\n    def __init__(self,\n                 filepath,\n                 frequency=10,\n                 monitor='val_loss',\n                 verbose=0,\n                 save_best_only=False,\n                 save_weights_only=False,\n                 mode='auto',\n                 options=None,\n                 **kwargs):\n        super().__init__(\n            filepath, \n            monitor, \n            verbose, \n            save_best_only, \n            save_weights_only,\n            mode, \n            \"epoch\", \n            options\n        )\n        self.epochs_since_last_save = 0\n        self.frequency = frequency\n\n    def on_epoch_end(self, epoch, logs=None):\n        self.epochs_since_last_save += 1\n        \n        if self.epochs_since_last_save % self.frequency == 0:\n            self._save_model(epoch=epoch, batch=None, logs=logs)\n\n    def on_train_batch_end(self, batch, logs=None):\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-10-31T01:55:43.863384Z","iopub.execute_input":"2022-10-31T01:55:43.863769Z","iopub.status.idle":"2022-10-31T01:55:45.074984Z","shell.execute_reply.started":"2022-10-31T01:55:43.863727Z","shell.execute_reply":"2022-10-31T01:55:45.073852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_path = \"/kaggle/working/training/cp.ckpt\"\ncp_callback = EpochModelCheckpoint(filepath=checkpoint_path, save_weights_only=True, verbose=1)\n\nmodel = get_model(checkpoint_path)\nepochs = 10\n\nhistory = model.fit(\n                  x_train,\n                  y_train,\n                  epochs=epochs,\n                  callbacks=[cp_callback],\n                  validation_data=(x_test, y_test)\n            )","metadata":{"execution":{"iopub.status.busy":"2022-10-31T02:01:36.411651Z","iopub.execute_input":"2022-10-31T02:01:36.412163Z","iopub.status.idle":"2022-10-31T02:16:40.877163Z","shell.execute_reply.started":"2022-10-31T02:01:36.412122Z","shell.execute_reply":"2022-10-31T02:16:40.875606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], \"r-\", label=\"loss\")\nplt.plot(history.history[\"val_loss\"], \"b-.\", label=\"validation loss\")\nplt.xlabel(\"iterations\")\nplt.ylabel(\"loss\")\nplt.legend(loc=\"upper left\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T02:16:45.543560Z","iopub.execute_input":"2022-10-31T02:16:45.544542Z","iopub.status.idle":"2022-10-31T02:16:45.835117Z","shell.execute_reply.started":"2022-10-31T02:16:45.544492Z","shell.execute_reply":"2022-10-31T02:16:45.833748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nMaking predictions on the test set","metadata":{}},{"cell_type":"code","source":"# A summary function of all the operations on the df\ndef data_pipeline(df):\n    df[\"dist_ball_A_goal\"] = calc_dist(\n        df[\"ball_pos_x\"], \n        df[\"ball_pos_y\"], \n        df[\"ball_pos_z\"], \n        A_goal_coords[0], \n        A_goal_coords[1], \n        A_goal_coords[2]\n    )\n\n    df[\"dist_ball_B_goal\"] = calc_dist(\n        df[\"ball_pos_x\"], \n        df[\"ball_pos_y\"], \n        df[\"ball_pos_z\"], \n        B_goal_coords[0], \n        B_goal_coords[1], \n        B_goal_coords[2]\n    )\n\n\n    players = [\"p0\", \"p1\", \"p2\", \"p3\", \"p4\", \"p5\"]\n    for p in players:\n        df[f\"{p}_dist_to_A_net\"] = calc_dist(\n            df[f\"{p}_pos_x\"], \n            df[f\"{p}_pos_y\"],\n            df[f\"{p}_pos_z\"], \n            A_goal_coords[0], \n            A_goal_coords[1], \n            A_goal_coords[2]\n        )\n\n        df[f\"{p}_dist_to_B_net\"] = calc_dist(\n            df[f\"{p}_pos_x\"], \n            df[f\"{p}_pos_y\"], \n            df[f\"{p}_pos_z\"], \n            B_goal_coords[0], \n            B_goal_coords[1], \n            B_goal_coords[2]\n        )\n\n    # Normalize data\n    # Min-Max Normalization\n\n    pos_x_cols = [col for col in df.columns if \"pos_x\" in col]\n    pos_y_cols = [col for col in df.columns if \"pos_y\" in col]\n\n    df[pos_x_cols] = (df[pos_x_cols]  - POS_X_MIN) / (POS_X_MAX - POS_X_MIN)\n    df[pos_y_cols] = (df[pos_y_cols]  - POS_Y_MIN) / (POS_Y_MAX - POS_Y_MIN)\n    \n    # NaN features\n    players = [\"p0\", \"p1\", \"p2\", \"p3\", \"p4\", \"p5\"]\n    for i, player in enumerate(players):\n        player_columns = player_features[i]\n        df[f\"{player}_na\"] = df[player_columns].isnull().any(axis=1).astype(\"float32\")\n    \n\n    # Note the RuntimeWarning gets raised because of the dtype we're using\n    fill = df.select_dtypes(include='number').dropna().mean().to_dict()\n\n    # Boost feature still has NaN\n    for k,v in fill.items():\n        if np.isnan(v):\n            fill[k] = 0\n\n    df.fillna(fill, inplace=True)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-31T02:16:54.745205Z","iopub.execute_input":"2022-10-31T02:16:54.746257Z","iopub.status.idle":"2022-10-31T02:16:54.760595Z","shell.execute_reply.started":"2022-10-31T02:16:54.746213Z","shell.execute_reply":"2022-10-31T02:16:54.759308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dtypes = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/test_dtypes.csv\")\ntest_dtypes = dict(test_dtypes.to_records(index=False))\n    \ntest_df = pd.read_csv(\"/kaggle/input/tabular-playground-series-oct-2022/test.csv\", dtype=test_dtypes)\ntest_df = data_pipeline(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T02:16:55.353298Z","iopub.execute_input":"2022-10-31T02:16:55.353699Z","iopub.status.idle":"2022-10-31T02:17:08.270865Z","shell.execute_reply.started":"2022-10-31T02:16:55.353667Z","shell.execute_reply":"2022-10-31T02:17:08.269564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_predict = test_df.drop(columns=[\"id\"])\npredictions = model.predict(x_predict)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T02:18:39.369260Z","iopub.execute_input":"2022-10-31T02:18:39.369693Z","iopub.status.idle":"2022-10-31T02:19:20.973963Z","shell.execute_reply.started":"2022-10-31T02:18:39.369655Z","shell.execute_reply":"2022-10-31T02:19:20.972830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(predictions, columns=[\"team_A_scoring_within_10sec\",\"team_B_scoring_within_10sec\"])\nsubmission.index.name = \"id\"\nsubmission.to_csv(\"submission.csv\", index=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T02:19:20.975498Z","iopub.execute_input":"2022-10-31T02:19:20.975851Z","iopub.status.idle":"2022-10-31T02:19:22.875384Z","shell.execute_reply.started":"2022-10-31T02:19:20.975821Z","shell.execute_reply":"2022-10-31T02:19:22.874115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Thank you for reading!\n\nThank you to the **SSA Google Machine Learning Bootcamp** for the awesome program as well! 🚀🚀🚀","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}