{"cells":[{"cell_type":"code","execution_count":1,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:23:28.441032Z","iopub.status.busy":"2023-02-06T16:23:28.440027Z","iopub.status.idle":"2023-02-06T16:23:31.600891Z","shell.execute_reply":"2023-02-06T16:23:31.599979Z","shell.execute_reply.started":"2023-02-06T16:23:28.440951Z"},"trusted":true},"outputs":[],"source":"import time\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.metrics import matthews_corrcoef\nimport torch, torchvision\nimport torch.nn as nn\nimport xgboost as xgb\nfrom xgboost import XGBRegressor\nfrom sklearn import preprocessing\nfrom torch import Tensor\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom torch.utils.data.dataloader import default_collate\nfrom sklearn import preprocessing\nimport gc"},{"cell_type":"code","execution_count":2,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:23:31.603568Z","iopub.status.busy":"2023-02-06T16:23:31.602467Z","iopub.status.idle":"2023-02-06T16:23:56.223297Z","shell.execute_reply":"2023-02-06T16:23:56.222066Z","shell.execute_reply.started":"2023-02-06T16:23:31.603533Z"},"trusted":true},"outputs":[{"ename":"FileNotFoundError","evalue":"[Errno 2] No such file or directory: '/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv'","output_type":"error","traceback":["\u001b[31m---------------------------------------------------------------------------\u001b[39m","\u001b[31mFileNotFoundError\u001b[39m                         Traceback (most recent call last)","\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[2]\u001b[39m\u001b[32m, line 3\u001b[39m\n\u001b[32m      1\u001b[39m data_kaggle = \u001b[33m\"\u001b[39m\u001b[33m/kaggle/input/nfl-player-contact-detection/\u001b[39m\u001b[33m\"\u001b[39m\n\u001b[32m      2\u001b[39m data = \u001b[33m'\u001b[39m\u001b[33m'\u001b[39m\n\u001b[32m----> \u001b[39m\u001b[32m3\u001b[39m train_tracking = \u001b[43mpd\u001b[49m\u001b[43m.\u001b[49m\u001b[43mread_csv\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43mf\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[38;5;132;43;01m{\u001b[39;49;00m\u001b[43mdata_kaggle\u001b[49m\u001b[38;5;132;43;01m}\u001b[39;49;00m\u001b[33;43mtrain_player_tracking.csv\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m)\u001b[49m\n\u001b[32m      4\u001b[39m train_helmets = pd.read_csv(\u001b[33mf\u001b[39m\u001b[33m\"\u001b[39m\u001b[38;5;132;01m{\u001b[39;00mdata_kaggle\u001b[38;5;132;01m}\u001b[39;00m\u001b[33mtrain_baseline_helmets.csv\u001b[39m\u001b[33m\"\u001b[39m)\n\u001b[32m      5\u001b[39m train_labels = pd.read_csv(\u001b[33mf\u001b[39m\u001b[33m\"\u001b[39m\u001b[38;5;132;01m{\u001b[39;00mdata_kaggle\u001b[38;5;132;01m}\u001b[39;00m\u001b[33mtrain_labels.csv\u001b[39m\u001b[33m\"\u001b[39m)\n","\u001b[36mFile \u001b[39m\u001b[32m~\\AppData\\Local\\Packages\\PythonSoftwareFoundation.Python.3.11_qbz5n2kfra8p0\\LocalCache\\local-packages\\Python311\\site-packages\\pandas\\io\\parsers\\readers.py:1026\u001b[39m, in \u001b[36mread_csv\u001b[39m\u001b[34m(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, delim_whitespace, low_memory, memory_map, float_precision, storage_options, dtype_backend)\u001b[39m\n\u001b[32m   1013\u001b[39m kwds_defaults = _refine_defaults_read(\n\u001b[32m   1014\u001b[39m     dialect,\n\u001b[32m   1015\u001b[39m     delimiter,\n\u001b[32m   (...)\u001b[39m\u001b[32m   1022\u001b[39m     dtype_backend=dtype_backend,\n\u001b[32m   1023\u001b[39m )\n\u001b[32m   1024\u001b[39m kwds.update(kwds_defaults)\n\u001b[32m-> \u001b[39m\u001b[32m1026\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[43m_read\u001b[49m\u001b[43m(\u001b[49m\u001b[43mfilepath_or_buffer\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43mkwds\u001b[49m\u001b[43m)\u001b[49m\n","\u001b[36mFile \u001b[39m\u001b[32m~\\AppData\\Local\\Packages\\PythonSoftwareFoundation.Python.3.11_qbz5n2kfra8p0\\LocalCache\\local-packages\\Python311\\site-packages\\pandas\\io\\parsers\\readers.py:620\u001b[39m, in \u001b[36m_read\u001b[39m\u001b[34m(filepath_or_buffer, kwds)\u001b[39m\n\u001b[32m    617\u001b[39m _validate_names(kwds.get(\u001b[33m\"\u001b[39m\u001b[33mnames\u001b[39m\u001b[33m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m))\n\u001b[32m    619\u001b[39m \u001b[38;5;66;03m# Create the parser.\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m620\u001b[39m parser = \u001b[43mTextFileReader\u001b[49m\u001b[43m(\u001b[49m\u001b[43mfilepath_or_buffer\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43m*\u001b[49m\u001b[43m*\u001b[49m\u001b[43mkwds\u001b[49m\u001b[43m)\u001b[49m\n\u001b[32m    622\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m chunksize \u001b[38;5;129;01mor\u001b[39;00m iterator:\n\u001b[32m    623\u001b[39m     \u001b[38;5;28;01mreturn\u001b[39;00m parser\n","\u001b[36mFile \u001b[39m\u001b[32m~\\AppData\\Local\\Packages\\PythonSoftwareFoundation.Python.3.11_qbz5n2kfra8p0\\LocalCache\\local-packages\\Python311\\site-packages\\pandas\\io\\parsers\\readers.py:1620\u001b[39m, in \u001b[36mTextFileReader.__init__\u001b[39m\u001b[34m(self, f, engine, **kwds)\u001b[39m\n\u001b[32m   1617\u001b[39m     \u001b[38;5;28mself\u001b[39m.options[\u001b[33m\"\u001b[39m\u001b[33mhas_index_names\u001b[39m\u001b[33m\"\u001b[39m] = kwds[\u001b[33m\"\u001b[39m\u001b[33mhas_index_names\u001b[39m\u001b[33m\"\u001b[39m]\n\u001b[32m   1619\u001b[39m \u001b[38;5;28mself\u001b[39m.handles: IOHandles | \u001b[38;5;28;01mNone\u001b[39;00m = \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[32m-> \u001b[39m\u001b[32m1620\u001b[39m \u001b[38;5;28mself\u001b[39m._engine = \u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43m_make_engine\u001b[49m\u001b[43m(\u001b[49m\u001b[43mf\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43mengine\u001b[49m\u001b[43m)\u001b[49m\n","\u001b[36mFile \u001b[39m\u001b[32m~\\AppData\\Local\\Packages\\PythonSoftwareFoundation.Python.3.11_qbz5n2kfra8p0\\LocalCache\\local-packages\\Python311\\site-packages\\pandas\\io\\parsers\\readers.py:1880\u001b[39m, in \u001b[36mTextFileReader._make_engine\u001b[39m\u001b[34m(self, f, engine)\u001b[39m\n\u001b[32m   1878\u001b[39m     \u001b[38;5;28;01mif\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m mode:\n\u001b[32m   1879\u001b[39m         mode += \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m\n\u001b[32m-> \u001b[39m\u001b[32m1880\u001b[39m \u001b[38;5;28mself\u001b[39m.handles = \u001b[43mget_handle\u001b[49m\u001b[43m(\u001b[49m\n\u001b[32m   1881\u001b[39m \u001b[43m    \u001b[49m\u001b[43mf\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1882\u001b[39m \u001b[43m    \u001b[49m\u001b[43mmode\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1883\u001b[39m \u001b[43m    \u001b[49m\u001b[43mencoding\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mencoding\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mNone\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1884\u001b[39m \u001b[43m    \u001b[49m\u001b[43mcompression\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mcompression\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mNone\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1885\u001b[39m \u001b[43m    \u001b[49m\u001b[43mmemory_map\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mmemory_map\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mFalse\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1886\u001b[39m \u001b[43m    \u001b[49m\u001b[43mis_text\u001b[49m\u001b[43m=\u001b[49m\u001b[43mis_text\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1887\u001b[39m \u001b[43m    \u001b[49m\u001b[43merrors\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mencoding_errors\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mstrict\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1888\u001b[39m \u001b[43m    \u001b[49m\u001b[43mstorage_options\u001b[49m\u001b[43m=\u001b[49m\u001b[38;5;28;43mself\u001b[39;49m\u001b[43m.\u001b[49m\u001b[43moptions\u001b[49m\u001b[43m.\u001b[49m\u001b[43mget\u001b[49m\u001b[43m(\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43mstorage_options\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;28;43;01mNone\u001b[39;49;00m\u001b[43m)\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m   1889\u001b[39m \u001b[43m\u001b[49m\u001b[43m)\u001b[49m\n\u001b[32m   1890\u001b[39m \u001b[38;5;28;01massert\u001b[39;00m \u001b[38;5;28mself\u001b[39m.handles \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[32m   1891\u001b[39m f = \u001b[38;5;28mself\u001b[39m.handles.handle\n","\u001b[36mFile \u001b[39m\u001b[32m~\\AppData\\Local\\Packages\\PythonSoftwareFoundation.Python.3.11_qbz5n2kfra8p0\\LocalCache\\local-packages\\Python311\\site-packages\\pandas\\io\\common.py:873\u001b[39m, in \u001b[36mget_handle\u001b[39m\u001b[34m(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)\u001b[39m\n\u001b[32m    868\u001b[39m \u001b[38;5;28;01melif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(handle, \u001b[38;5;28mstr\u001b[39m):\n\u001b[32m    869\u001b[39m     \u001b[38;5;66;03m# Check whether the filename is to be opened in binary mode.\u001b[39;00m\n\u001b[32m    870\u001b[39m     \u001b[38;5;66;03m# Binary mode does not support 'encoding' and 'newline'.\u001b[39;00m\n\u001b[32m    871\u001b[39m     \u001b[38;5;28;01mif\u001b[39;00m ioargs.encoding \u001b[38;5;129;01mand\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m ioargs.mode:\n\u001b[32m    872\u001b[39m         \u001b[38;5;66;03m# Encoding\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m873\u001b[39m         handle = \u001b[38;5;28;43mopen\u001b[39;49m\u001b[43m(\u001b[49m\n\u001b[32m    874\u001b[39m \u001b[43m            \u001b[49m\u001b[43mhandle\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m    875\u001b[39m \u001b[43m            \u001b[49m\u001b[43mioargs\u001b[49m\u001b[43m.\u001b[49m\u001b[43mmode\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m    876\u001b[39m \u001b[43m            \u001b[49m\u001b[43mencoding\u001b[49m\u001b[43m=\u001b[49m\u001b[43mioargs\u001b[49m\u001b[43m.\u001b[49m\u001b[43mencoding\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m    877\u001b[39m \u001b[43m            \u001b[49m\u001b[43merrors\u001b[49m\u001b[43m=\u001b[49m\u001b[43merrors\u001b[49m\u001b[43m,\u001b[49m\n\u001b[32m    878\u001b[39m \u001b[43m            \u001b[49m\u001b[43mnewline\u001b[49m\u001b[43m=\u001b[49m\u001b[33;43m\"\u001b[39;49m\u001b[33;43m\"\u001b[39;49m\u001b[43m,\u001b[49m\n\u001b[32m    879\u001b[39m \u001b[43m        \u001b[49m\u001b[43m)\u001b[49m\n\u001b[32m    880\u001b[39m     \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[32m    881\u001b[39m         \u001b[38;5;66;03m# Binary mode\u001b[39;00m\n\u001b[32m    882\u001b[39m         handle = \u001b[38;5;28mopen\u001b[39m(handle, ioargs.mode)\n","\u001b[31mFileNotFoundError\u001b[39m: [Errno 2] No such file or directory: '/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv'"]}],"source":"data_kaggle = \"/kaggle/input/nfl-player-contact-detection/\"\ndata = ''\ntrain_tracking = pd.read_csv(f\"{data_kaggle}train_player_tracking.csv\")\ntrain_helmets = pd.read_csv(f\"{data_kaggle}train_baseline_helmets.csv\")\ntrain_labels = pd.read_csv(f\"{data_kaggle}train_labels.csv\")"},{"cell_type":"markdown","metadata":{},"source":"# UTILS"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:23:56.225199Z","iopub.status.busy":"2023-02-06T16:23:56.224885Z","iopub.status.idle":"2023-02-06T16:23:56.244025Z","shell.execute_reply":"2023-02-06T16:23:56.242969Z","shell.execute_reply.started":"2023-02-06T16:23:56.225171Z"},"trusted":true},"outputs":[],"source":"def joindfs(tracking, helmets, labels, training=True):\n    fps = 59.94\n    frame_delta = 6\n\n    tracking = tracking.copy()\n    helmets = helmets.copy()\n    labels = labels.copy()\n\n    df = labels[['contact_id', 'contact']]\n\n    # taking the values required from the label df and converting steps to frames\n    df = df.copy()\n    df[\"game_play\"] = df.contact_id.apply(lambda x: \"_\".join(x.split(\"_\")[0:2]))\n    df[\"step\"] = df.contact_id.apply(lambda x: x.split(\"_\")[2]).astype(int)\n    df[\"nfl_player_id_1\"] = df.contact_id.apply(lambda x: x.split(\"_\")[3]).astype(int)\n    df[\"nfl_player_id_2\"] = df.contact_id.apply(lambda x: x.split(\"_\")[4])\n    df[\"nfl_player_id_2\"] = df[\"nfl_player_id_2\"].apply(lambda x: -1 if x == \"G\" else x).astype(int)\n    \n\n    # Columns to take from the tracking df, you can add speed and acceleration if you want but remember to rename the columns below\n    tracking_columns = [\"nfl_player_id\", \"x_position\", \"y_position\", \"game_play\", \"step\", 'team', 'orientation', 'direction', 'acceleration', 'sa', 'speed', 'distance']\n\n    # tracking df for each player\n    df_tracking_1 = tracking[tracking_columns]\n    df_tracking_1 = df_tracking_1.rename(columns={\"nfl_player_id\": \"nfl_player_id_1\", \"x_position\": \"x_position_1\", \"y_position\": \"y_position_1\", 'team': 'team_1', 'orientation': 'orientation_1', 'direction': 'direction_1', 'acceleration': 'acceleration_1', 'sa': 'sa_1', 'speed':'speed_1', 'distance': 'distance_1'})\n    df_tracking_2 = tracking[tracking_columns]\n    df_tracking_2 = df_tracking_2.rename(columns={\"nfl_player_id\": \"nfl_player_id_2\", \"x_position\": \"x_position_2\", \"y_position\": \"y_position_2\", 'team': 'team_2', 'orientation': 'orientation_2', 'direction': 'direction_2', 'acceleration': 'acceleration_2', 'sa': 'sa_2', 'speed':'speed_2', 'distance': 'distance_2'})\n\n    # merge the label df with the tracking ones for each player\n    df = pd.merge(df_tracking_1, df, how=\"right\", on=['nfl_player_id_1', 'game_play', 'step'])\n    df = pd.merge(df_tracking_2, df, how=\"right\", on=['nfl_player_id_2', 'game_play', 'step'])\n    \n    df[\"frame\"] = df[\"step\"].apply(lambda x: 300 + int(x * 0.1 * fps / frame_delta) * frame_delta)\n    \n    df['same_team'] = np.where(df['team_1'] == df['team_2'], 1, -1)\n    df['rel_orientation'] = df['orientation_1'] - df['orientation_2']\n    df['rel_direction'] = df['direction_1'] - df['direction_2']\n\n    # calculate the player to player distance\n    df[\"distance\"] = ((df.x_position_2 - df.x_position_1) ** 2 + (df.y_position_2 - df.y_position_1) ** 2) ** 0.5\n\n    helmets_columns = ['game_play', 'view', 'frame', 'nfl_player_id', 'left', 'width', 'top', 'height']\n    helmets = helmets.astype({'left': 'int32', 'width': 'int32', 'top': 'int32', 'height': 'int32'})\n    \n    df_helmets_1 = helmets[helmets_columns]\n    df_helmets_1 = df_helmets_1.rename(columns={\"nfl_player_id\": \"nfl_player_id_1\", \"left\": \"left_1\", \"width\": \"width_1\", \"top\": \"top_1\", \"height\": \"height_1\"})\n    df_helmets_2 = helmets[helmets_columns]\n    df_helmets_2 = df_helmets_2.rename(columns={\"nfl_player_id\": \"nfl_player_id_2\", \"left\": \"left_2\", \"width\": \"width_2\", \"top\": \"top_2\", \"height\": \"height_2\"})\n\n    df = pd.merge(df_helmets_1, df, how=\"right\", on=['nfl_player_id_1', 'game_play', 'frame'])\n    df = pd.merge(df_helmets_2, df, how=\"right\", on=['nfl_player_id_2', 'game_play', 'frame', 'view'])\n    df['view'] = preprocessing.LabelEncoder().fit_transform(df['view'])\n    \n    #df = df.groupby('contact_id', as_index=False).mean()\n    #df[\"game_play\"] = df.contact_id.apply(lambda x: \"_\".join(x.split(\"_\")[0:2]))\n    #df = pd.get_dummies(df, columns=['view'])\n    # uncomment this if you are dealing with both at the same time\n    return df\n\n    df_player = df[df.nfl_player_id_2 != -1]\n    \n    df_ground = df[df.nfl_player_id_2 == -1]\n    df_ground = df_ground.drop(labels=['left_2', 'width_2', 'top_2', 'height_2', 'x_position_2', 'y_position_2', 'distance', 'distance_2'], axis=1)\n    \n    \n\n    return df_player, df_ground"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:23:56.246971Z","iopub.status.busy":"2023-02-06T16:23:56.246536Z","iopub.status.idle":"2023-02-06T16:23:56.265395Z","shell.execute_reply":"2023-02-06T16:23:56.264215Z","shell.execute_reply.started":"2023-02-06T16:23:56.24694Z"},"trusted":true},"outputs":[],"source":"def sample_contacts(data_frame, n_samples_per_contact):\n    # Calculate the total number of contacts in the DataFrame\n    total_contacts = data_frame['contact'].sum()\n    \n    # Extract rows with contact = 1 (contacts) and contact = 0 (non-contacts)\n    contacts = data_frame[data_frame['contact'] == 1]\n    non_contacts = data_frame[data_frame['contact'] == 0]\n    \n    # Sample n_samples_per_contact times the number of non-contacts to create a balanced dataset\n    sampled_non_contacts = non_contacts.sample(n_samples_per_contact * total_contacts)\n    \n    # Concatenate the sampled contacts and non-contacts to create the final DataFrame\n    sampled_data_frame = pd.concat([contacts, sampled_non_contacts])\n    \n    return sampled_data_frame"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:23:56.267112Z","iopub.status.busy":"2023-02-06T16:23:56.266639Z","iopub.status.idle":"2023-02-06T16:23:56.286537Z","shell.execute_reply":"2023-02-06T16:23:56.284888Z","shell.execute_reply.started":"2023-02-06T16:23:56.267074Z"},"trusted":true},"outputs":[],"source":"def train_test_split(df, feature_column, n=20, test_percent=0.2):\n    # Initialize empty DataFrames for training and testing sets\n    train, test = pd.DataFrame(), pd.DataFrame()\n\n    # Get the unique game plays from the DataFrame\n    plays = df['game_play'].unique()\n    length_plays = len(plays)\n    print(length_plays)\n\n    # Determine the index at which to split the data into training and testing sets\n    train_split = length_plays - int(length_plays * test_percent)\n\n    # Populate the training set with data for plays before the split\n    for play in plays[:train_split]:\n        df_play = df[df['game_play'] == play]\n        train = pd.concat([train, df_play])\n\n    # Populate the testing set with data for plays after the split\n    for play in plays[train_split:length_plays]:\n        df_play = df[df['game_play'] == play]\n        test = pd.concat([test, df_play])\n\n    # If a testing set is required (test_percent > 0), sample contacts in the training set\n    if test_percent > 0:\n        train = sample_contacts(train, n)\n        return train[feature_column], train['contact'], test[feature_column], test['contact']\n    else:\n        # If no testing set is required, return the training set as is\n        return train, train['contact']"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:23:56.288952Z","iopub.status.busy":"2023-02-06T16:23:56.288499Z","iopub.status.idle":"2023-02-06T16:24:42.78349Z","shell.execute_reply":"2023-02-06T16:24:42.781645Z","shell.execute_reply.started":"2023-02-06T16:23:56.288917Z"},"trusted":true},"outputs":[],"source":"# Joining three DataFrames: train_tracking, train_helmets, and train_labels\n# to create a single DataFrame named 'df'.\ndf = joindfs(train_tracking, train_helmets, train_labels)\n\n# List of feature column names that will be used for training the model\nFEATURES = ['distance', 'frame', 'step', 'left_1', 'width_1', 'top_1', 'height_1', \n            'step', 'left_2', 'width_2', 'top_2', 'height_2', 'view', 'same_team', \n            'rel_orientation', 'orientation_1', 'orientation_2', 'rel_direction', \n            'direction_1', 'direction_2', 'speed_1', 'speed_2', 'acceleration_1', \n            'acceleration_2', 'sa_1', 'sa_2', 'x_position_1', 'y_position_1', \n            'x_position_2', 'y_position_2', 'distance_1', 'distance_2']\n\n# Split the 'df' DataFrame into training and testing sets with the specified feature columns.\n# 'X_df' contains the feature columns, and 'y_train' contains the target variable.\nX_df, y_train = train_test_split(df, FEATURES, test_percent=0)\n\n# Extract the feature columns from 'X_df' and fill any missing values with 0.\nX_train = X_df[FEATURES]\nX_train.fillna(0, inplace=True)\n\n# Perform garbage collection to free up memory.\ngc.collect()\n"},{"cell_type":"markdown","metadata":{},"source":"# PLAYER MODEL"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:27:58.017635Z","iopub.status.busy":"2023-02-06T16:27:58.01716Z","iopub.status.idle":"2023-02-06T16:27:59.958827Z","shell.execute_reply":"2023-02-06T16:27:59.95768Z","shell.execute_reply.started":"2023-02-06T16:27:58.017603Z"},"trusted":true},"outputs":[],"source":"# Create a list 'cols' containing column names that end with '_1' but are not 'nfl_player_id_1'.\ncols = [i[:-2] for i in X_train.columns if i[-2:] == '_1' and i != 'nfl_player_id_1']\n\n# Create a new DataFrame 'new_x_train' by taking the absolute difference between corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_diff' suffix.\nnew_x_train = pd.DataFrame(np.abs(X_train[[i + '_1' for i in cols]].values - X_train[[i + '_2' for i in cols]].values),\n                          columns=[i + '_diff' for i in cols])\n\n# Add the new '_diff' columns to the 'X_train' DataFrame.\nX_train[[i + '_diff' for i in cols]] = new_x_train\n\n# Create a list 'cols' containing column names without the '_1' or '_2' suffix.\ncols = ['x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration', 'sa']\n\n# Create a new DataFrame 'new_x_train' by taking the element-wise product of corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_prod' suffix.\nnew_x_train = pd.DataFrame(X_train[[i + '_1' for i in cols]].values * X_train[[i + '_2' for i in cols]].values,\n                          columns=[i + '_prod' for i in cols])\n\n# Add the new '_prod' columns to the 'X_train' DataFrame.\nX_train[[i + '_prod' for i in cols]] = new_x_train\ngc.collect();"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:28:01.523551Z","iopub.status.busy":"2023-02-06T16:28:01.523111Z","iopub.status.idle":"2023-02-06T16:28:01.530149Z","shell.execute_reply":"2023-02-06T16:28:01.528292Z","shell.execute_reply.started":"2023-02-06T16:28:01.523513Z"},"trusted":true},"outputs":[],"source":"# Create an instance of the XGBoost classifier\nCLF = xgb.XGBClassifier()"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T16:28:01.532212Z","iopub.status.busy":"2023-02-06T16:28:01.531812Z","iopub.status.idle":"2023-02-06T17:02:28.211262Z","shell.execute_reply":"2023-02-06T17:02:28.210425Z","shell.execute_reply.started":"2023-02-06T16:28:01.532176Z"},"trusted":true},"outputs":[],"source":"# Fit the XGBoost classifier (CLF) to the training data\nCLF.fit(X_train.values,  # Features of the training data\n         y_train.values,  # Labels of the training data\n         eval_set=[(X_train.values, y_train.values)],  # Evaluation set for monitoring model performance\n         verbose=1)  # Verbosity level (1 for displaying training progress)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:28.213341Z","iopub.status.busy":"2023-02-06T17:02:28.212262Z","iopub.status.idle":"2023-02-06T17:02:28.226665Z","shell.execute_reply":"2023-02-06T17:02:28.225872Z","shell.execute_reply.started":"2023-02-06T17:02:28.213282Z"},"trusted":true},"outputs":[],"source":"print(CLF.feature_importances_)\ngc.collect();"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:28.426136Z","iopub.status.busy":"2023-02-06T17:02:28.425633Z","iopub.status.idle":"2023-02-06T17:02:44.488757Z","shell.execute_reply":"2023-02-06T17:02:44.486581Z","shell.execute_reply.started":"2023-02-06T17:02:28.4261Z"},"trusted":true},"outputs":[],"source":"test_pred = CLF.predict(X_train.values)\nprint(matthews_corrcoef(y_train, test_pred))"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:44.491287Z","iopub.status.busy":"2023-02-06T17:02:44.490873Z","iopub.status.idle":"2023-02-06T17:02:44.712106Z","shell.execute_reply":"2023-02-06T17:02:44.711022Z","shell.execute_reply.started":"2023-02-06T17:02:44.49125Z"},"trusted":true},"outputs":[],"source":"test_tracking = pd.read_csv(f\"{data_kaggle}test_player_tracking.csv\")\ntest_helmets = pd.read_csv(f\"{data_kaggle}test_baseline_helmets.csv\")\ntest_labels = pd.read_csv(f\"{data_kaggle}sample_submission.csv\")"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:44.714196Z","iopub.status.busy":"2023-02-06T17:02:44.713434Z","iopub.status.idle":"2023-02-06T17:02:45.178097Z","shell.execute_reply":"2023-02-06T17:02:45.17707Z","shell.execute_reply.started":"2023-02-06T17:02:44.714136Z"},"trusted":true},"outputs":[],"source":"test_df = joindfs(test_tracking, test_helmets, test_labels)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.179787Z","iopub.status.busy":"2023-02-06T17:02:45.179408Z","iopub.status.idle":"2023-02-06T17:02:45.247355Z","shell.execute_reply":"2023-02-06T17:02:45.246152Z","shell.execute_reply.started":"2023-02-06T17:02:45.179755Z"},"trusted":true},"outputs":[],"source":"X_test_df, y_test = train_test_split(test_df, FEATURES, test_percent=0)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.249331Z","iopub.status.busy":"2023-02-06T17:02:45.249027Z","iopub.status.idle":"2023-02-06T17:02:45.271629Z","shell.execute_reply":"2023-02-06T17:02:45.269627Z","shell.execute_reply.started":"2023-02-06T17:02:45.249305Z"},"trusted":true},"outputs":[],"source":"X_test = X_test_df[FEATURES]"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.274263Z","iopub.status.busy":"2023-02-06T17:02:45.273791Z","iopub.status.idle":"2023-02-06T17:02:45.285914Z","shell.execute_reply":"2023-02-06T17:02:45.284873Z","shell.execute_reply.started":"2023-02-06T17:02:45.274224Z"},"trusted":true},"outputs":[],"source":"X_test.fillna(0, inplace=True)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.292735Z","iopub.status.busy":"2023-02-06T17:02:45.292296Z","iopub.status.idle":"2023-02-06T17:02:45.366641Z","shell.execute_reply":"2023-02-06T17:02:45.365007Z","shell.execute_reply.started":"2023-02-06T17:02:45.292677Z"},"trusted":true},"outputs":[],"source":"# Create a list 'cols' containing column names that end with '_1' but are not 'nfl_player_id_1'.\ncols = [i[:-2] for i in X_test.columns if i[-2:] == '_1' and i != 'nfl_player_id_1']\n\n# Create a new DataFrame 'new_x_train' by taking the absolute difference between corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_diff' suffix.\nnew_x_test = pd.DataFrame(np.abs(X_test[[i + '_1' for i in cols]].values - X_test[[i + '_2' for i in cols]].values),\n                          columns=[i + '_diff' for i in cols])\n\n# Add the new '_diff' columns to the 'X_test' DataFrame.\nX_test[[i + '_diff' for i in cols]] = new_x_test\n\n# Create a list 'cols' containing column names without the '_1' or '_2' suffix.\ncols = ['x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration', 'sa']\n\n# Create a new DataFrame 'new_x_train' by taking the element-wise product of corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_prod' suffix.\nnew_x_test = pd.DataFrame(X_test[[i + '_1' for i in cols]].values * X_test[[i + '_2' for i in cols]].values,\n                          columns=[i + '_prod' for i in cols])\n\n# Add the new '_prod' columns to the 'X_test' DataFrame.\nX_test[[i + '_prod' for i in cols]] = new_x_test\ngc.collect();"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.368644Z","iopub.status.busy":"2023-02-06T17:02:45.368324Z","iopub.status.idle":"2023-02-06T17:02:45.539072Z","shell.execute_reply":"2023-02-06T17:02:45.537672Z","shell.execute_reply.started":"2023-02-06T17:02:45.368617Z"},"trusted":true},"outputs":[],"source":"predictions = CLF.predict(X_test.values)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.541397Z","iopub.status.busy":"2023-02-06T17:02:45.541029Z","iopub.status.idle":"2023-02-06T17:02:45.56153Z","shell.execute_reply":"2023-02-06T17:02:45.559233Z","shell.execute_reply.started":"2023-02-06T17:02:45.541367Z"},"trusted":true},"outputs":[],"source":"sample_test = pd.DataFrame()\nsample_test['contact_id'] = X_test_df['contact_id']\nsample_test['contact'] = predictions "},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.565262Z","iopub.status.busy":"2023-02-06T17:02:45.564225Z","iopub.status.idle":"2023-02-06T17:02:45.634368Z","shell.execute_reply":"2023-02-06T17:02:45.633349Z","shell.execute_reply.started":"2023-02-06T17:02:45.5652Z"},"trusted":true},"outputs":[],"source":"sample_test = sample_test.groupby('contact_id', as_index=False).mean()"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.635969Z","iopub.status.busy":"2023-02-06T17:02:45.635657Z","iopub.status.idle":"2023-02-06T17:02:45.645215Z","shell.execute_reply":"2023-02-06T17:02:45.642608Z","shell.execute_reply.started":"2023-02-06T17:02:45.635945Z"},"trusted":true},"outputs":[],"source":"test_labels.set_index('contact_id', inplace=True)\nsample_test.set_index('contact_id', inplace=True)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.65031Z","iopub.status.busy":"2023-02-06T17:02:45.649326Z","iopub.status.idle":"2023-02-06T17:02:45.694933Z","shell.execute_reply":"2023-02-06T17:02:45.693272Z","shell.execute_reply.started":"2023-02-06T17:02:45.650241Z"},"trusted":true},"outputs":[],"source":"sample_test = sample_test.reindex(test_labels.index)\nsample_test.reset_index(inplace=True)\nsample_test['contact'] = (sample_test.contact > 0.5).astype(int)"},{"cell_type":"code","execution_count":null,"metadata":{"execution":{"iopub.execute_input":"2023-02-06T17:02:45.698424Z","iopub.status.busy":"2023-02-06T17:02:45.697895Z","iopub.status.idle":"2023-02-06T17:02:45.774121Z","shell.execute_reply":"2023-02-06T17:02:45.77253Z","shell.execute_reply.started":"2023-02-06T17:02:45.698372Z"},"trusted":true},"outputs":[],"source":"sample_test.to_csv(\"/kaggle/working/submission.csv\", index=False)"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.9"}},"nbformat":4,"nbformat_minor":4}