{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![A cute cat](https://media.cnn.com/api/v1/images/stellar/prod/230121121737-01-nfl-playoffs-preview.jpg?c=16x9&q=h_720,w_1280,c_fill)\n\n","metadata":{}},{"cell_type":"markdown","source":"The goal of this notebook is to detect external contact experienced by players during an NFL football game. I will only player tracking data to identify moments with contact to help improve player safety.","metadata":{}},{"cell_type":"markdown","source":"# STARTING : Importing necessary libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom sklearn.preprocessing import StandardScaler\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, roc_auc_score, make_scorer\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV, cross_val_score\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import plot_confusion_matrix\nimport lightgbm as lgb\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\n\n# Ignore all warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:30:36.942112Z","iopub.execute_input":"2023-03-03T09:30:36.944200Z","iopub.status.idle":"2023-03-03T09:30:36.952799Z","shell.execute_reply.started":"2023-03-03T09:30:36.944127Z","shell.execute_reply":"2023-03-03T09:30:36.951618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DATA","metadata":{}},{"cell_type":"code","source":"data_path = Path(\"/kaggle/input/nfl-player-contact-detection\")\ntrain_labels = pd.read_csv(data_path / \"train_labels.csv\", parse_dates=['datetime'])\ntrain_tracking = pd.read_csv(data_path / \"train_player_tracking.csv\", parse_dates=['datetime'])\ndata_path = Path(\"/kaggle/input/nfl-player-contact-detection\")\nsub = pd.read_csv(data_path / \"sample_submission.csv\")\ntest_tracking = pd.read_csv(data_path / \"test_player_tracking.csv\", parse_dates=['datetime'])","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:14:30.044157Z","iopub.execute_input":"2023-03-03T08:14:30.044689Z","iopub.status.idle":"2023-03-03T08:15:16.638952Z","shell.execute_reply.started":"2023-03-03T08:14:30.044641Z","shell.execute_reply":"2023-03-03T08:15:16.637284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:24:38.150268Z","iopub.execute_input":"2023-03-03T08:24:38.151578Z","iopub.status.idle":"2023-03-03T08:24:38.168219Z","shell.execute_reply.started":"2023-03-03T08:24:38.151523Z","shell.execute_reply":"2023-03-03T08:24:38.166636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:26:40.056087Z","iopub.execute_input":"2023-03-03T08:26:40.056869Z","iopub.status.idle":"2023-03-03T08:26:40.088173Z","shell.execute_reply.started":"2023-03-03T08:26:40.056818Z","shell.execute_reply":"2023-03-03T08:26:40.086510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The train labels data is a set of *EVERY COMBINATION OF TWO PLAYERS & 1 PLAYER & GROUND* that has **4.721.618**  rows and 6 columns + the target variable (contact). 3 of features are integers, 1 is datetime and 3 are objects. The features with a short descriptions are following:\n* **contact_id:** A combination of the game_play, player_ids and step columns.\n* **game_play**: the unique ID for the game and play.\n* **nfl_player_id_1** The lower numbered player id in the contact pair. If contact with ground then this is just the player id.\n* **nfl_player_id_2**: The larger number player id in the contact pair. If for contact with the ground, this will contain an uppercase \"G\"\n* **step**: A number representing each each timestep for each play, starting at 0 at the moment of the play starting, and incrementing by 1 every 0.1 seconds.\n* **datetime**: The timetamp of the contact, at 10Hz\n* **contact**: Whether contact occurred","metadata":{}},{"cell_type":"code","source":"train_tracking.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:24:41.183871Z","iopub.execute_input":"2023-03-03T08:24:41.184405Z","iopub.status.idle":"2023-03-03T08:24:41.213831Z","shell.execute_reply.started":"2023-03-03T08:24:41.184359Z","shell.execute_reply":"2023-03-03T08:24:41.211762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tracking.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:26:51.491042Z","iopub.execute_input":"2023-03-03T08:26:51.492255Z","iopub.status.idle":"2023-03-03T08:26:51.761886Z","shell.execute_reply.started":"2023-03-03T08:26:51.492206Z","shell.execute_reply":"2023-03-03T08:26:51.760337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The train player tracking data has **1.353.053** rows for EACH PLAYER and 17 features. 8 of the features are floats, 5 are integers, 1 is datetime and 3 are objects. The features with a short descriptions are following:\n* **game_play**: Unique game key and play id combination for the play.\n* **game_key**: the ID code for the game.\n* **play_id**: the ID code for the play.\n* **nfl_player_id**: the player's ID code.\n* **datetime**: timestamp at 10 Hz.\n* **step**: timestep within play relative to the play start.\n* **position**: the football position of the player.\n* **team**: team of the player, either home or away.\n* **jersey_number**: Player jersey number\n* **x_position**: player position along the long axis of the field. See figure below.\n* **y_position**: player position along the short axis of the field. See figure below.\n* **speed**: speed in yards/second.\n* **distance**: distance traveled from prior time point, in yards.\n* **orientation**: orientation of player (deg).\n* **direction**: angle of player motion (deg).\n* **acceleration**: magnitiude of the total acceleration in yards/second^2.\n* **sa**: Signed acceleration yards/second^2 in the direction the player is moving.","metadata":{}},{"cell_type":"markdown","source":"# DATA PREPROCESSING","metadata":{}},{"cell_type":"markdown","source":"##  Step 1: ***process_train_data ()*** function:\nThis function combines train labels & train player tracking in player-wise & player-ground-wise level.","metadata":{}},{"cell_type":"code","source":"def process_train_data(train_labels, train_tracking):\n    games = train_labels\n    games[['gameId', 'playId']] = games['game_play'].str.split('_', 1, expand=True)\n    games = games[['gameId', 'playId', 'step', 'nfl_player_id_1', 'nfl_player_id_2', 'contact']]\n    train_tracking = train_tracking.rename(columns={'game_key':'gameId', 'play_id':'playId', 'nfl_player_id':'nfl_player_id_1'})\n    train_tracking['gameId'] = train_tracking['gameId'].astype(int)\n    games['gameId'] = games['gameId'].astype(int)\n    games['playId'] = games['playId'].astype(int)\n    games['nfl_player_id_1'] = games['nfl_player_id_1'].astype(int)\n    train_tracking = train_tracking[train_tracking.step>=0]\n    games = games.merge(train_tracking[['gameId', 'playId', 'step', 'nfl_player_id_1', 'x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration', 'distance', 'sa']], on=['gameId', 'playId', 'step', 'nfl_player_id_1'], how='left')\n    games = games.rename(columns={'x_position':'x_position_player_id_1', 'y_position':'y_position_player_id_1'})\n    ground_games = games[games.nfl_player_id_2=='G']\n    games = games[games.nfl_player_id_2!='G']\n    train_tracking = train_tracking.rename(columns={'nfl_player_id_1':'nfl_player_id_2'})\n    games['nfl_player_id_2'] = games['nfl_player_id_2'].astype(int)\n    games = games.merge(train_tracking[['gameId', 'playId', 'step', 'nfl_player_id_2', 'x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration', 'distance', 'sa']], on=['gameId', 'playId', 'step', 'nfl_player_id_2'], how='left')\n    games = games.rename(columns={'x_position': 'x_position_player_id_2', 'y_position': 'y_position_player_id_2'})\n    ground_games['x_position_player_id_2'] = 0\n    ground_games['y_position_player_id_2'] = 0\n    return games, ground_games\ntrain_games, train_ground_games=process_train_data(train_labels, train_tracking)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:39:50.353031Z","iopub.execute_input":"2023-03-03T08:39:50.353986Z","iopub.status.idle":"2023-03-03T08:40:10.665454Z","shell.execute_reply.started":"2023-03-03T08:39:50.353907Z","shell.execute_reply":"2023-03-03T08:40:10.664374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_games.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:36:42.378124Z","iopub.execute_input":"2023-03-03T08:36:42.378944Z","iopub.status.idle":"2023-03-03T08:36:42.409769Z","shell.execute_reply.started":"2023-03-03T08:36:42.378882Z","shell.execute_reply":"2023-03-03T08:36:42.408331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ground_games.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:36:55.033441Z","iopub.execute_input":"2023-03-03T08:36:55.033915Z","iopub.status.idle":"2023-03-03T08:36:55.057273Z","shell.execute_reply.started":"2023-03-03T08:36:55.033874Z","shell.execute_reply":"2023-03-03T08:36:55.055941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Step 2: ***process_test_data ()*** function:\nThis function combines sample submission & test player tracking in player-wise & player-ground-wise level.","metadata":{}},{"cell_type":"code","source":"def process_test_data(sub, test_tracking):\n    sub[['gameId', 'playId', 'step', 'nfl_player_id_1', 'nfl_player_id_2']]=sub['contact_id'].str.split('_', 5, expand=True)\n    test_tracking = test_tracking.rename(columns={'game_key':'gameId', 'play_id':'playId', 'nfl_player_id':'nfl_player_id_1'})\n    test_tracking['gameId'] = test_tracking['gameId'].astype(int)\n    sub['gameId'] = sub['gameId'].astype(int)\n    sub['playId'] = sub['playId'].astype(int)\n    sub['step'] = sub['step'].astype(int)\n    sub['nfl_player_id_1'] = sub['nfl_player_id_1'].astype(int)\n    test_tracking=test_tracking[test_tracking.step>=0]\n    games=sub.merge(test_tracking[['gameId', 'playId', 'step', 'nfl_player_id_1', 'x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration','distance', 'sa']], on=['gameId', 'playId', 'step', 'nfl_player_id_1'], how='left')\n    games = games.rename(columns={'x_position':'x_position_player_id_1', 'y_position':'y_position_player_id_1'})\n    ground_games = games[games.nfl_player_id_2=='G']\n    games = games[games.nfl_player_id_2!='G']\n    test_tracking = test_tracking.rename(columns={'nfl_player_id_1':'nfl_player_id_2'})\n    games['nfl_player_id_2'] = games['nfl_player_id_2'].astype(int)\n    games = games.merge(test_tracking[['gameId', 'playId', 'step', 'nfl_player_id_2', 'x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration','distance' ,'sa']], on=['gameId', 'playId', 'step', 'nfl_player_id_2'], how='left')\n    games = games.rename(columns={'x_position': 'x_position_player_id_2', 'y_position': 'y_position_player_id_2'})\n    ground_games['x_position_player_id_2'] = 0\n    ground_games['y_position_player_id_2'] = 0\n    games = pd.concat([games, ground_games])\n    return games, ground_games\ntest_games, test_ground_games=process_test_data(sub, test_tracking)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:42:01.347131Z","iopub.execute_input":"2023-03-03T08:42:01.347711Z","iopub.status.idle":"2023-03-03T08:42:01.608578Z","shell.execute_reply.started":"2023-03-03T08:42:01.347662Z","shell.execute_reply":"2023-03-03T08:42:01.607147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_games.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:44:06.947033Z","iopub.execute_input":"2023-03-03T08:44:06.947579Z","iopub.status.idle":"2023-03-03T08:44:06.979589Z","shell.execute_reply.started":"2023-03-03T08:44:06.947528Z","shell.execute_reply":"2023-03-03T08:44:06.978473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ground_games.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:44:15.194877Z","iopub.execute_input":"2023-03-03T08:44:15.196611Z","iopub.status.idle":"2023-03-03T08:44:15.219789Z","shell.execute_reply.started":"2023-03-03T08:44:15.196537Z","shell.execute_reply":"2023-03-03T08:44:15.218354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE ENGINEERING","metadata":{}},{"cell_type":"code","source":"train_games.sort_values(['gameId', 'playId', 'nfl_player_id_1', 'nfl_player_id_2', 'step'], inplace=True)\ntrain_games.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:45:52.045715Z","iopub.execute_input":"2023-03-03T08:45:52.046516Z","iopub.status.idle":"2023-03-03T08:45:54.934618Z","shell.execute_reply.started":"2023-03-03T08:45:52.046454Z","shell.execute_reply":"2023-03-03T08:45:54.933126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Distance between players","metadata":{}},{"cell_type":"code","source":"train_games['distances'] = np.sqrt((train_games['x_position_player_id_1'] - train_games['x_position_player_id_2'])**2 + (train_games['y_position_player_id_1'] - train_games['y_position_player_id_1'])**2)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:46:54.338283Z","iopub.execute_input":"2023-03-03T08:46:54.339590Z","iopub.status.idle":"2023-03-03T08:46:57.440784Z","shell.execute_reply.started":"2023-03-03T08:46:54.339536Z","shell.execute_reply":"2023-03-03T08:46:57.439615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Mean Distance btw players: calculates mean distances of each combination of players during the play.","metadata":{}},{"cell_type":"code","source":"means=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['distances'].mean().reset_index().rename(columns={'distances':'meandistanceBTWPlayers'})\ntrain_games=train_games.merge(means, on=['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'], how='left')\nmin_distance=train_games.loc[train_games.groupby(['nfl_player_id_1'])[\"meandistanceBTWPlayers\"].idxmin()][['gameId', 'playId', 'nfl_player_id_1', 'nfl_player_id_2', 'meandistanceBTWPlayers']]\nmin_distance.rename(columns={'meandistanceBTWPlayers':'minDistance'}, inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"instance_1=train_games[(train_games.gameId==58168)&(train_games.playId==3392)]\ninstance_1.groupby(['nfl_player_id_1','nfl_player_id_2']).distances.mean().reset_index().rename(columns={'distances': 'meandistanceBTWPlayers'}).head(21)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T08:54:15.693283Z","iopub.execute_input":"2023-03-03T08:54:15.693776Z","iopub.status.idle":"2023-03-03T08:54:15.735368Z","shell.execute_reply.started":"2023-03-03T08:54:15.693738Z","shell.execute_reply":"2023-03-03T08:54:15.733543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems like during the play combination of player 37084 was the most close with player 43854","metadata":{}},{"cell_type":"markdown","source":"* Relative orientation, direction and acceleration:","metadata":{}},{"cell_type":"code","source":"train_games['rel_orient'] =abs(train_games['orientation_x'] - train_games['orientation_y'])\ntrain_games['rel_speed'] = abs(train_games['speed_x'] - train_games['speed_y'])\ntrain_games['rel_direction'] = abs(train_games['direction_x'] - train_games['direction_y'])\ntrain_games['rel_acceleration'] = abs(train_games['acceleration_x'] - train_games['acceleration_y'])","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:08:19.801166Z","iopub.execute_input":"2023-03-03T09:08:19.801690Z","iopub.status.idle":"2023-03-03T09:08:20.042318Z","shell.execute_reply.started":"2023-03-03T09:08:19.801646Z","shell.execute_reply":"2023-03-03T09:08:20.040460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Avg speed and distance travelled:","metadata":{}},{"cell_type":"code","source":"train_games['avg_speed'] = (train_games['speed_x'] + train_games['speed_x']) / 2\ntrain_games['avg_progressed_dis'] = (train_games['distance_x'] + train_games['distance_y']) / 2","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Changes of speed, distance, acceleration and sa for each steps:","metadata":{}},{"cell_type":"code","source":"train_games['speedChange_x']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speed_x'].pct_change()\ntrain_games['speedChange_y']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speed_y'].pct_change()\ntrain_games['distanceChange']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['distances'].pct_change()\ntrain_games['distanceChange_x']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['distance_x'].pct_change()\ntrain_games['distanceChange_y']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['distance_y'].pct_change()\ntrain_games['accelerationChange_x']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['acceleration_x'].pct_change()\ntrain_games['accelerationChange_y']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['acceleration_y'].pct_change()\ntrain_games['saChange_x']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['sa_x'].pct_change()\ntrain_games['saChange_y']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['sa_y'].pct_change()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:01:55.847850Z","iopub.execute_input":"2023-03-03T09:01:55.848365Z","iopub.status.idle":"2023-03-03T09:02:07.275097Z","shell.execute_reply.started":"2023-03-03T09:01:55.848321Z","shell.execute_reply":"2023-03-03T09:02:07.273758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Checking whether the change in distance and change in speed are in opposite directions:","metadata":{}},{"cell_type":"code","source":"train_games.loc[train_games['distanceChange']*train_games['speedChange_x']<0, 'DisSpeedDir']=-1\ntrain_games.loc[train_games['distanceChange']*train_games['speedChange_x']>0, 'DisSpeedDir']=1","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:02:10.216960Z","iopub.execute_input":"2023-03-03T09:02:10.217730Z","iopub.status.idle":"2023-03-03T09:02:10.475547Z","shell.execute_reply.started":"2023-03-03T09:02:10.217683Z","shell.execute_reply":"2023-03-03T09:02:10.473643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Rolling means to reveal underlying trends:","metadata":{}},{"cell_type":"code","source":"train_games['speed_xRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speed_x'].rolling(window=4).mean().reset_index().speed_x\ntrain_games['distancesRM']= train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['distances'].rolling(window=3).mean().reset_index().distances\ntrain_games['speedChange_xRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speedChange_x'].rolling(window=3).mean().reset_index().speedChange_x\ntrain_games['speedChange_yRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speedChange_y'].rolling(window=3).mean().reset_index().speedChange_y\ntrain_games['speed_yRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speed_y'].rolling(window=4).mean().reset_index().speed_y\ntrain_games['accelerationChange_xRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['accelerationChange_x'].rolling(window=3).mean().reset_index().accelerationChange_x\ntrain_games['accelerationChange_yRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['accelerationChange_y'].rolling(window=3).mean().reset_index().accelerationChange_y\ntrain_games['acceleration_xRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['acceleration_x'].rolling(window=3).mean().reset_index().acceleration_x\ntrain_games['acceleration_yRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['acceleration_y'].rolling(window=3).mean().reset_index().acceleration_y\ntrain_games['sa_xRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['sa_x'].rolling(window=3).mean().reset_index().sa_x\ntrain_games['sa_yRM']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['sa_y'].rolling(window=3).mean().reset_index().sa_y","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:05:51.681800Z","iopub.execute_input":"2023-03-03T09:05:51.682377Z","iopub.status.idle":"2023-03-03T09:06:51.567410Z","shell.execute_reply.started":"2023-03-03T09:05:51.682325Z","shell.execute_reply":"2023-03-03T09:06:51.566030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Convert directions and orientations to radians\n* Calculate the velocities & accelerations in x and y directions for each player:","metadata":{}},{"cell_type":"code","source":"# Convert directions and orientations to radians\ntheta1 = np.radians(train_games['orientation_x'])\ntheta2 = np.radians(train_games['orientation_y'])\nphi1 = np.radians(train_games['direction_x'])\nphi2 = np.radians(train_games['direction_y'])\n\n# Calculate the velocities in x and y directions for each player\nv1_x = train_games['speed_x'] * np.cos(phi1)\nv1_y = train_games['speed_x'] * np.sin(phi1)\nv2_x = train_games['speed_y'] * np.cos(phi2)\nv2_y = train_games['speed_y'] * np.sin(phi2)\n\n# Calculate the acceleration in x and y directions for each player\nacc1_x = train_games['acceleration_x'] * np.cos(phi1)\nacc1_y = train_games['acceleration_x'] * np.sin(phi1)\nacc2_x = train_games['acceleration_y'] * np.cos(phi2)\nacc2_y = train_games['acceleration_y'] * np.sin(phi2)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:11:05.864639Z","iopub.execute_input":"2023-03-03T09:11:05.865129Z","iopub.status.idle":"2023-03-03T09:11:06.348746Z","shell.execute_reply.started":"2023-03-03T09:11:05.865083Z","shell.execute_reply":"2023-03-03T09:11:06.346704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Calculate the relative velocity & acceleration vector components\n","metadata":{}},{"cell_type":"code","source":"# Calculate the relative velocity vector components\nv_rel_x = v2_x - v1_x\nv_rel_y = v2_y - v1_y\n\n# Calculate the relative acceleration vector components\nacc_rel_x = acc2_x - acc1_x\nacc_rel_y = acc2_y - acc1_y","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Calculates the magnitude of the relative velocity & acceleration vector:","metadata":{}},{"cell_type":"code","source":"# Calculate the magnitude of the relative velocity vector\nv_rel_mag = np.sqrt(v_rel_x**2 + v_rel_y**2)\n\n# Calculate the magnitude of the relative acceleration vector\nacc_rel_mag = np.sqrt(acc_rel_x**2 + acc_rel_y**2)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:15:14.811426Z","iopub.execute_input":"2023-03-03T09:15:14.812155Z","iopub.status.idle":"2023-03-03T09:15:14.960504Z","shell.execute_reply.started":"2023-03-03T09:15:14.812111Z","shell.execute_reply":"2023-03-03T09:15:14.958681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Calculates expected distance between players**\n* vx1_rel =v1_x+v_rel_x: This line calculates the relative x-velocity of player 1 with respect to player 2 by adding the x-component of player 1's velocity, v1_x, to the relative x-component of the velocity between player 1 and player 2, v_rel_x.\n\n* vy1_rel =v1_y+v_rel_y: This line calculates the relative y-velocity of player 1 with respect to player 2 by adding the y-component of player 1's velocity, v1_y, to the relative y-component of the velocity between player 1 and player 2, v_rel_y.\n\n* vx2_rel =v2_x+v_rel_x: This line calculates the relative x-velocity of player 2 with respect to player 1 by adding the x-component of player 2's velocity, v2_x, to the relative x-component of the velocity between player 1 and player 2, v_rel_x.\n\n* vy2_rel =v2_y+v_rel_y: This line calculates the relative y-velocity of player 2 with respect to player 1 by adding the y-component of player 2's velocity, v2_y, to the relative y-component of the velocity between player 1 and player 2, v_rel_y.\n\n* x1_new = x_position_player_id_1 + vx1_rel: This line calculates the new x-coordinate of player 1's position by adding the relative x-velocity of player 1 to their current x-coordinate.\n\n* y1_new = y_position_player_id_2 + vy1_rel: This line calculates the new y-coordinate of player 1's position by adding the relative y-velocity of player 1 to their current y-coordinate.\n\n* x2_new = x_position_player_id_2 + vx2_rel: This line calculates the new x-coordinate of player 2's position by adding the relative x-velocity of player 2 to their current x-coordinate.\n\n* y2_new = y_position_player_id_2 + vy2_rel: This line calculates the new y-coordinate of player 2's position by adding the relative y-velocity of player 2 to their current y-coordinate.","metadata":{}},{"cell_type":"code","source":"x_position_player_id_1=train_games['x_position_player_id_1'].values\ny_position_player_id_1=train_games['y_position_player_id_1'].values\nx_position_player_id_2=train_games['x_position_player_id_2'].values\ny_position_player_id_2=train_games['y_position_player_id_2'].values\n\n\nvx1_rel =v1_x+v_rel_x\nvy1_rel =v1_y+v_rel_y\n\nvx2_rel =v2_x+v_rel_x\nvy2_rel =v2_y+v_rel_y\n\nx1_new = x_position_player_id_1 + vx1_rel\ny1_new = y_position_player_id_2 + vy1_rel\n\nx2_new = x_position_player_id_2 + vx2_rel\ny2_new = y_position_player_id_2 + vy2_rel\n\ntrain_games['x1_new']=x1_new\ntrain_games['y1_new']=y1_new\ntrain_games['x2_new']=x2_new\ntrain_games['y2_new']=y2_new\ntrain_games['newdistances'] = np.sqrt((train_games['x1_new'] - train_games['x2_new'])**2 + (train_games['y1_new'] - train_games['y2_new'])**2)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:23:46.472473Z","iopub.execute_input":"2023-03-03T09:23:46.473065Z","iopub.status.idle":"2023-03-03T09:23:46.849437Z","shell.execute_reply.started":"2023-03-03T09:23:46.473014Z","shell.execute_reply":"2023-03-03T09:23:46.848074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Percent change of \"newdistances\"","metadata":{}},{"cell_type":"code","source":"train_games['newdistancesChg']=train_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['newdistances'].pct_change()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:24:54.046735Z","iopub.execute_input":"2023-03-03T09:24:54.047215Z","iopub.status.idle":"2023-03-03T09:24:55.193806Z","shell.execute_reply.started":"2023-03-03T09:24:54.047166Z","shell.execute_reply":"2023-03-03T09:24:55.192385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Calculates differences between consecutive values of the x and y coordinates of the two players.","metadata":{}},{"cell_type":"code","source":"train_games['x_position_player_id_1_diff'] = train_games['x_position_player_id_1'].diff()\ntrain_games['y_position_player_id_1_diff'] = train_games['x_position_player_id_1'].diff()\ntrain_games['x_position_player_id_2_diff'] = train_games['x_position_player_id_2'].diff()\ntrain_games['y_position_player_id_2_diff'] = train_games['y_position_player_id_2'].diff()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:27:42.888788Z","iopub.execute_input":"2023-03-03T09:27:42.889283Z","iopub.status.idle":"2023-03-03T09:27:43.037745Z","shell.execute_reply.started":"2023-03-03T09:27:42.889240Z","shell.execute_reply":"2023-03-03T09:27:43.035565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_games=train_games.fillna(0)\ntrain_games=train_games[~train_games.isin([np.nan, np.inf, -np.inf]).any(1)]","metadata":{"execution":{"iopub.status.busy":"2023-03-03T09:28:43.842005Z","iopub.execute_input":"2023-03-03T09:28:43.842509Z","iopub.status.idle":"2023-03-03T09:28:54.887234Z","shell.execute_reply.started":"2023-03-03T09:28:43.842469Z","shell.execute_reply":"2023-03-03T09:28:54.885256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE SELECTION WITH RFECV","metadata":{}},{"cell_type":"markdown","source":"RFECV stands for Recursive Feature Elimination with Cross-Validation. It is a feature selection technique in machine learning that is used to select the most important features from a given dataset.\n\nThe algorithm works by recursively removing features from the dataset and evaluating the performance of the model using the remaining features. It then selects the optimal number of features by cross-validating the model and selecting the subset of features that produces the best results.\n\nIn RFECV, a base estimator (such as a decision tree or a linear regression model) is trained on the entire dataset, and the feature importances are calculated. The least important features are then pruned from the dataset, and the process is repeated until the desired number of features is reached.","metadata":{}},{"cell_type":"code","source":"# Separate the features and target variable\nX = train_games.drop(columns=['contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2'], axis=1)\ny = train_games['contact']\n\n# Split the data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nfrom sklearn.feature_selection import RFECV\n\n# define the LightGBM classifier\nclf = lgb.LGBMClassifier()\n\n# define the feature selector\nselector = RFECV(clf, step=1, cv=5, scoring='roc_auc')\n\n# perform feature selection on the training data\nselector.fit(X_train, y_train)\n\n# get the selected features\nselected_features = X_train.columns[selector.support_]\n\n# train the LightGBM classifier using the selected features\nclf.fit(X_train[selected_features], y_train)\n\n# make predictions on the test data using the trained classifier\ny_pred = clf.predict_proba(X_test[selected_features])[:, 1]\n\n# calculate the AUC score\nauc_score = roc_auc_score(y_test, y_pred)\n\n# print the selected features and the AUC score\nprint(\"Selected features:\", selected_features)\nprint(\"AUC score:\", auc_score)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T11:49:52.952158Z","iopub.execute_input":"2023-03-03T11:49:52.952667Z","iopub.status.idle":"2023-03-03T14:54:24.145446Z","shell.execute_reply.started":"2023-03-03T11:49:52.952621Z","shell.execute_reply":"2023-03-03T14:54:24.143005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LGBM MODEL \n## a) MODEL & EVALUATION ON TRAIN & VALIDATION SET\nThis model will be implemented for two player contact cases (train_games & test_games)","metadata":{}},{"cell_type":"code","source":"train=train_games[['contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2',\n'y_position_player_id_1', 'x_position_player_id_2',\n'y_position_player_id_2', 'distances', 'meandistanceBTWPlayers',\n'speed_xRM', 'rel_orient', \"newdistances\", \"x_position_player_id_1_diff\", \n\"x_position_player_id_2_diff\", \"y_position_player_id_2_diff\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-03T15:29:37.729946Z","iopub.execute_input":"2023-03-03T15:29:37.730610Z","iopub.status.idle":"2023-03-03T15:29:38.602318Z","shell.execute_reply.started":"2023-03-03T15:29:37.730557Z","shell.execute_reply":"2023-03-03T15:29:38.600349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import classification_report, confusion_matrix\n\n# Separate the features and target variable\nX = train.drop(columns=['contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2'], axis=1)\ny = train['contact']\n\n# Split the data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define the LightGBM classifier\nparams = {\n  'objective': 'binary',\n    'metric': 'auc',\n    'boosting_type': 'gbdt',\n    'num_leaves': 150,\n    'learning_rate': 0.1,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.9,\n    'bagging_freq': 2,\n    'verbose': 0,\n    'min_data_in_leaf': 100,\n    'lambda_l1': 0.1,\n    'lambda_l2': 0.1,\n    'scale_pos_weight': 4.0, \n}\n\n# Train the LightGBM classifier on the training data\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_test = lgb.Dataset(X_test, y_test, reference=lgb_train)\nmodel = lgb.train(params, lgb_train, num_boost_round=1000, valid_sets=[lgb_train, lgb_test], early_stopping_rounds=10, verbose_eval=50)\n\n# Evaluate the model on the testing data\ny_pred = model.predict(X_test)\ny_pred_class = np.round(y_pred)\nprint(confusion_matrix(y_test, y_pred_class))\nprint(classification_report(y_test, y_pred_class))\n","metadata":{"execution":{"iopub.status.busy":"2023-03-03T15:29:41.003893Z","iopub.execute_input":"2023-03-03T15:29:41.004479Z","iopub.status.idle":"2023-03-03T15:45:38.459240Z","shell.execute_reply.started":"2023-03-03T15:29:41.004421Z","shell.execute_reply":"2023-03-03T15:45:38.457363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The model's performance looks quite good overall. Here is a breakdown of the key performance metrics:\n\n* ***Precision***: Precision measures how often the model correctly predicts a positive label (in this case, label 1). In this case, the precision for label 1 is 0.88, which means that out of all the instances the model predicted as positive, 88% of them were actually positive.\n\n* ***Recall***: Recall measures how often the model correctly identifies positive labels. In this case, the recall for label 1 is 0.89, which means that out of all the actual positive instances, 89% of them were correctly identified by the model.\n\n* ***F1-score***: F1-score is a weighted average of precision and recall. In this case, the F1-score for label 1 is 0.89, which means that the model's precision and recall are both quite good for label 1.\n\n* ***Accuracy***: Accuracy measures the overall percentage of correct predictions made by the model. In this case, the accuracy is very high at 1.00, which means that the model is making very few mistakes overall.\n\n* ***Macro-average***: The macro-average is an average of the precision, recall, and F1-score for both labels. In this case, the macro-average for both labels is 0.94, which means that the model performs quite well on average.\n\n* ***Weighted-average***: The weighted-average is an average of the precision, recall, and F1-score for both labels, weighted by the number of instances in each label. In this case, the weighted-average for both labels is 1.00, which means that the model performs very well overall.\n\nBased on these metrics, it seems that the model is performing well on this particular dataset. ","metadata":{}},{"cell_type":"code","source":"model.save_model('model.txt')","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:05:31.000320Z","iopub.execute_input":"2023-03-03T16:05:31.001435Z","iopub.status.idle":"2023-03-03T16:05:31.253771Z","shell.execute_reply.started":"2023-03-03T16:05:31.001274Z","shell.execute_reply":"2023-03-03T16:05:31.251587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## b) PREDICTIONS ON TEST SET\nWe're gonna predict test_games contacts.","metadata":{}},{"cell_type":"markdown","source":"Feature engineering:","metadata":{}},{"cell_type":"code","source":"test_games=test_games[test_games.nfl_player_id_2!='G']\ntest_games.sort_values(['gameId', 'playId', 'nfl_player_id_1', 'nfl_player_id_2', 'step'], inplace=True)\n# calculates the Euclidean distance \ntest_games['distances'] = np.sqrt((test_games['x_position_player_id_1'] - test_games['x_position_player_id_2'])**2 + (test_games['y_position_player_id_1'] - test_games['y_position_player_id_1'])**2)\nmeans=test_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['distances'].mean().reset_index().rename(columns={'distances':'meandistanceBTWPlayers'})\ntest_games=test_games.merge(means, on=['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'], how='left')\nmin_distance=test_games.loc[test_games.groupby(['nfl_player_id_1'])[\"meandistanceBTWPlayers\"].idxmin()][['gameId', 'playId', 'nfl_player_id_1', 'nfl_player_id_2', 'meandistanceBTWPlayers']]\n\n\ntest_games['speed_xRM']=test_games.groupby(['gameId','playId', 'nfl_player_id_1', 'nfl_player_id_2'])['speed_x'].rolling(window=4).mean().reset_index().speed_x\ntest_games['rel_orient'] = abs(test_games['orientation_x']-test_games['orientation_y'])\n\n\nx_position_player_id_1=test_games['x_position_player_id_1'].values\ny_position_player_id_1=test_games['y_position_player_id_1'].values\nx_position_player_id_2=test_games['x_position_player_id_2'].values\ny_position_player_id_2=test_games['y_position_player_id_2'].values\n\n# Convert directions and orientations to radians\ntheta1 = np.radians(test_games['orientation_x'])\ntheta2 = np.radians(test_games['orientation_y'])\nphi1 = np.radians(test_games['direction_x'])\nphi2 = np.radians(test_games['direction_y'])\n\n# Calculate the velocities in x and y directions for each player\nv1_x = test_games['speed_x'] * np.cos(phi1)\nv1_y = test_games['speed_x'] * np.sin(phi1)\nv2_x = test_games['speed_y'] * np.cos(phi2)\nv2_y = test_games['speed_y'] * np.sin(phi2)\n\n# Calculate the relative velocity vector components\nv_rel_x = v2_x - v1_x\nv_rel_y = v2_y - v1_y\n\nvx1_rel =v1_x+v_rel_x\nvy1_rel =v1_y+v_rel_y\n\nvx2_rel =v2_x+v_rel_x\nvy2_rel =v2_y+v_rel_y\n\nx1_new = x_position_player_id_1 + vx1_rel\ny1_new = y_position_player_id_2 + vy1_rel\n\nx2_new = x_position_player_id_2 + vx2_rel\ny2_new = y_position_player_id_2 + vy2_rel\n\ntest_games['x1_new']=x1_new\ntest_games['y1_new']=y1_new\ntest_games['x2_new']=x2_new\ntest_games['y2_new']=y2_new\ntest_games['newdistances'] = np.sqrt((test_games['x1_new'] - test_games['x2_new'])**2 + (test_games['y1_new'] - test_games['y2_new'])**2)\n\ntest_games['x_position_player_id_1_diff'] = test_games['x_position_player_id_1'].diff()\ntest_games['x_position_player_id_2_diff'] = test_games['x_position_player_id_2'].diff()\ntest_games['y_position_player_id_2_diff'] = test_games['y_position_player_id_2'].diff()\ntest=test_games.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:08:44.655248Z","iopub.execute_input":"2023-03-03T16:08:44.655874Z","iopub.status.idle":"2023-03-03T16:08:44.923472Z","shell.execute_reply.started":"2023-03-03T16:08:44.655824Z","shell.execute_reply":"2023-03-03T16:08:44.921406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=test_games[[\"contact_id\",'contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2',\n'y_position_player_id_1', 'x_position_player_id_2',\n'y_position_player_id_2', 'distances', 'meandistanceBTWPlayers',\n'speed_xRM', 'rel_orient', \"newdistances\", \"x_position_player_id_1_diff\", \n\"x_position_player_id_2_diff\", \"y_position_player_id_2_diff\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:09:17.114187Z","iopub.execute_input":"2023-03-03T16:09:17.114744Z","iopub.status.idle":"2023-03-03T16:09:17.132771Z","shell.execute_reply.started":"2023-03-03T16:09:17.114703Z","shell.execute_reply":"2023-03-03T16:09:17.130852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate the features and target variable\nX_test = test.drop(columns=['contact',\"contact_id\", 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2'], axis=1)\n# load your trained model\nmodel = lgb.Booster(model_file='model.txt')\n# make predictions on the unlabeled test data\npredictions = model.predict(X_test)\n# convert the predictions to binary class labels\npredictions_binary = (predictions > 0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:09:33.777338Z","iopub.execute_input":"2023-03-03T16:09:33.777841Z","iopub.status.idle":"2023-03-03T16:09:35.687743Z","shell.execute_reply.started":"2023-03-03T16:09:33.777798Z","shell.execute_reply":"2023-03-03T16:09:35.685574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission=test[[\"contact_id\"]]\nsample_submission['y_pred']=predictions_binary\nsample_submission.rename(columns={'y_pred':'contact'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:10:51.493861Z","iopub.execute_input":"2023-03-03T16:10:51.494389Z","iopub.status.idle":"2023-03-03T16:10:51.511562Z","shell.execute_reply.started":"2023-03-03T16:10:51.494344Z","shell.execute_reply":"2023-03-03T16:10:51.509846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Class 1 prediction ratio is .01","metadata":{}},{"cell_type":"code","source":"sample_submission.contact.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:11:38.988902Z","iopub.execute_input":"2023-03-03T16:11:38.989519Z","iopub.status.idle":"2023-03-03T16:11:39.010332Z","shell.execute_reply.started":"2023-03-03T16:11:38.989474Z","shell.execute_reply":"2023-03-03T16:11:39.008417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# c) PLAYER-GROUND PREDICTIONS ON TRAİN & VALIDATION SETS\nAnother LGBM model will be implemented on train/test_ground_games, but before that let's select features.","metadata":{}},{"cell_type":"code","source":"def create_features_grounds(df):\n    # FUTURES FOR train_ground_games and test_ground_games:\n    # Future 1: DISTANCE between TWO PLAYERS.\n    ground_games=df\n    ground_games=ground_games.sort_values(['gameId', 'playId', 'nfl_player_id_1', 'nfl_player_id_2', 'step'])\n    # Future 1: SPEED CHANGE of player\n    ground_games['speedChange']=ground_games.groupby(['gameId','playId', 'nfl_player_id_1'])['speed'].pct_change()\n    # Future 2: SPEED CHANGE DIRECTION of player\n    ground_games.loc[ground_games.speedChange>0,'speedDirection']= 1\n    ground_games.loc[ground_games.speedChange<0,'speedDirection']= -1\n    ground_games.loc[ground_games.speedChange==0,'speedDirection']= 0\n    # Future 3: ACCELERATION CHANGE of player\n    ground_games['accelerationChange']=ground_games.groupby(['gameId','playId', 'nfl_player_id_1'])['acceleration'].pct_change()\n    # Future 4: ACCELERATION CHANGE DIRECTION of player\n    ground_games.loc[ground_games.speedChange>0,'accelerationDirection']= 1\n    ground_games.loc[ground_games.speedChange<0,'accelerationDirection']= -1\n    ground_games.loc[ground_games.speedChange==0,'accelerationDirection']= 0\n    # Future 5: SPEED CHANGE RS of player\n    ground_games['speedChangeMA']=ground_games.groupby(['gameId', 'playId', 'nfl_player_id_1'])['speedChange'].rolling(8).mean().reset_index().speedChange\n    # Future 6: SPEED RS of player\n    ground_games['speedMA']=ground_games.groupby(['gameId', 'playId', 'nfl_player_id_1'])['speed'].rolling(8).mean().reset_index().speed\n    # Future 7: ACCELERATION CHANGE RS of player\n    ground_games['accChangeMA']=ground_games.groupby(['gameId', 'playId', 'nfl_player_id_1'])['accelerationChange'].rolling(8).mean().reset_index().accelerationChange\n    ground_games['accelerationDiff']=ground_games['acceleration'].diff()\n    return ground_games\n","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:17:28.715968Z","iopub.execute_input":"2023-03-03T16:17:28.716856Z","iopub.status.idle":"2023-03-03T16:17:28.735152Z","shell.execute_reply.started":"2023-03-03T16:17:28.716787Z","shell.execute_reply":"2023-03-03T16:17:28.732924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_grounds=create_features_grounds(train_ground_games)\ntrain_grounds=train_grounds.fillna(0)\ntrain_grounds=train_grounds[~train_grounds.isin([np.nan, np.inf, -np.inf]).any(1)]","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:17:45.720185Z","iopub.execute_input":"2023-03-03T16:17:45.720681Z","iopub.status.idle":"2023-03-03T16:17:47.720060Z","shell.execute_reply.started":"2023-03-03T16:17:45.720640Z","shell.execute_reply":"2023-03-03T16:17:47.718561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Selecting best features:","metadata":{}},{"cell_type":"code","source":"# Separate the features and target variable\nX = train_grounds.drop(columns=['contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2'], axis=1)\ny = train_grounds['contact']\n\n# Split the data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nfrom sklearn.feature_selection import RFECV\n\n# define the LightGBM classifier\nclf = lgb.LGBMClassifier()\n\n# define the feature selector\nselector = RFECV(clf, step=1, cv=5, scoring='roc_auc')\n\n# perform feature selection on the training data\nselector.fit(X_train, y_train)\n\n# get the selected features\nselected_features = X_train.columns[selector.support_]\n\n# train the LightGBM classifier using the selected features\nclf.fit(X_train[selected_features], y_train)\n\n# make predictions on the test data using the trained classifier\ny_pred = clf.predict_proba(X_test[selected_features])[:, 1]\n\n# calculate the AUC score\nauc_score = roc_auc_score(y_test, y_pred)\n\n# print the selected features and the AUC score\nprint(\"Selected features:\", selected_features)\nprint(\"AUC score:\", auc_score)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:20:01.510368Z","iopub.execute_input":"2023-03-03T16:20:01.510884Z","iopub.status.idle":"2023-03-03T16:25:32.544936Z","shell.execute_reply.started":"2023-03-03T16:20:01.510840Z","shell.execute_reply":"2023-03-03T16:25:32.543145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=train_grounds[['contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2',\n    'x_position_player_id_1', 'y_position_player_id_1', 'speed',\n       'direction', 'orientation', 'acceleration', 'distance', 'sa',\n       'speedChange', 'accelerationChange', 'accelerationDirection',\n       'speedChangeMA', 'speedMA', 'accChangeMA', 'accelerationDiff']]","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:55:44.986062Z","iopub.execute_input":"2023-03-03T16:55:44.987152Z","iopub.status.idle":"2023-03-03T16:55:45.033869Z","shell.execute_reply.started":"2023-03-03T16:55:44.987089Z","shell.execute_reply":"2023-03-03T16:55:45.032536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate the features and target variable\nX = train.drop(columns=['contact', 'gameId','playId','step','nfl_player_id_1','nfl_player_id_2'], axis=1)\ny = train['contact']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:55:47.882688Z","iopub.execute_input":"2023-03-03T16:55:47.883158Z","iopub.status.idle":"2023-03-03T16:55:48.013971Z","shell.execute_reply.started":"2023-03-03T16:55:47.883118Z","shell.execute_reply":"2023-03-03T16:55:48.012500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\n# Train the LightGBM classifier on the training data\nparams = {\n    'objective': 'binary',\n    'metric': 'auc',\n    'boosting_type': 'gbdt',\n    'num_leaves': 1500,\n    'learning_rate': 0.1,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.9,\n    'bagging_freq': 2,\n    'verbose': 0,\n    'min_data_in_leaf': 100,\n    'lambda_l1': 0.1,\n    'lambda_l2': 0.1,\n    'scale_pos_weight': 9.0, \n}\n\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_test = lgb.Dataset(X_test, y_test, reference=lgb_train)\nmodel2 = lgb.train(params, lgb_train, num_boost_round=1000, valid_sets=[lgb_train, lgb_test], early_stopping_rounds=10, verbose_eval=50)\n\n# Evaluate the model on the testing data\ny_pred = model2.predict(X_test)\ny_pred_class = np.round(y_pred)\nprint(confusion_matrix(y_test, y_pred_class))\nprint(classification_report(y_test, y_pred_class))\n\n# Save the trained model to a file\nmodel2.save_model('modelground.txt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Optimizing threshold:","metadata":{}},{"cell_type":"code","source":"# Predict probabilities for test data\ny_pred_prob = model2.predict(X_test)\n\n# Define a threshold\nthreshold = 0.3\n\n# Convert probabilities to predicted classes based on threshold\ny_pred_class = (y_pred_prob >= threshold).astype(int)\n\n# Print confusion matrix\nprint(confusion_matrix(y_test, y_pred_class))\nprint(classification_report(y_test, y_pred_class))","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:37:29.300158Z","iopub.execute_input":"2023-03-03T16:37:29.300703Z","iopub.status.idle":"2023-03-03T16:37:34.860195Z","shell.execute_reply.started":"2023-03-03T16:37:29.300658Z","shell.execute_reply":"2023-03-03T16:37:34.858146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The model's performance looks quite good overall. Here is a breakdown of the key performance metrics:\n\n* ***Precision***: Precision measures how often the model correctly predicts a positive label (in this case, label 1). In this case, the precision for label 1 is 0.90, which means that out of all the instances the model predicted as positive, 90% of them were actually positive.\n\n* ***Recall***: Recall measures how often the model correctly identifies positive labels. In this case, the recall for label 1 is 0.85, which means that out of all the actual positive instances, 85% of them were correctly identified by the model.\n\n* ***F1-score***: F1-score is a weighted average of precision and recall. In this case, the F1-score for label 1 is 0.88, which means that the model's precision and recall are both quite good for label 1.\n\n* ***Accuracy***: Accuracy measures the overall percentage of correct predictions made by the model. In this case, the accuracy is quite high at 0.99, which means that the model is making very few mistakes overall.\n\n* ***Macro-average***: The macro-average is an average of the precision, recall, and F1-score for both labels. In this case, the macro-average for both labels is 0.94, which means that the model performs quite well on average.\n\n* ***Weighted-average***: The weighted-average is an average of the precision, recall, and F1-score for both labels, weighted by the number of instances in each label. In this case, the weighted-average for both labels is 0.99, which means that the model performs very well overall.\n\nBased on these metrics, it seems that the model is performing well on this particular dataset.","metadata":{}},{"cell_type":"markdown","source":"# d) PLAYER-GROUND PREDICTIONS ON TEST SETS","metadata":{}},{"cell_type":"code","source":"test_grounds=create_features_grounds(test_ground_games)\ntest_grounds=test_grounds.fillna(0)\ntest_grounds=test_grounds[~test_grounds.isin([np.nan, np.inf, -np.inf]).any(1)]\ntest_grounds.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:56:22.844112Z","iopub.execute_input":"2023-03-03T16:56:22.844597Z","iopub.status.idle":"2023-03-03T16:56:22.957970Z","shell.execute_reply.started":"2023-03-03T16:56:22.844556Z","shell.execute_reply":"2023-03-03T16:56:22.956219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=test_grounds[['x_position_player_id_1', 'y_position_player_id_1', 'speed',\n       'direction', 'orientation', 'acceleration', 'distance', 'sa',\n       'speedChange', 'accelerationChange', 'accelerationDirection',\n       'speedChangeMA', 'speedMA', 'accChangeMA', 'accelerationDiff']]","metadata":{"execution":{"iopub.status.busy":"2023-03-03T16:57:59.925819Z","iopub.execute_input":"2023-03-03T16:57:59.926521Z","iopub.status.idle":"2023-03-03T16:57:59.936521Z","shell.execute_reply.started":"2023-03-03T16:57:59.926465Z","shell.execute_reply":"2023-03-03T16:57:59.934471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load your trained model\nmodel = lgb.Booster(model_file='modelground.txt')\n# make predictions on the unlabeled test data\npredictions = model.predict(test)\n# convert the predictions to binary class labels\npredictions_binary = (predictions > 0.3).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T17:01:17.908751Z","iopub.execute_input":"2023-03-03T17:01:17.909284Z","iopub.status.idle":"2023-03-03T17:01:18.444556Z","shell.execute_reply.started":"2023-03-03T17:01:17.909241Z","shell.execute_reply":"2023-03-03T17:01:18.442437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2023-03-03T17:00:32.894982Z","iopub.execute_input":"2023-03-03T17:00:32.895554Z","iopub.status.idle":"2023-03-03T17:00:32.938548Z","shell.execute_reply.started":"2023-03-03T17:00:32.895508Z","shell.execute_reply":"2023-03-03T17:00:32.936802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_grounds","metadata":{"execution":{"iopub.status.busy":"2023-03-03T17:00:43.176275Z","iopub.execute_input":"2023-03-03T17:00:43.176847Z","iopub.status.idle":"2023-03-03T17:00:43.231175Z","shell.execute_reply.started":"2023-03-03T17:00:43.176797Z","shell.execute_reply":"2023-03-03T17:00:43.228898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":".05% of prediction belongs to class 1.","metadata":{}},{"cell_type":"code","source":"sample_submission_grounds=test_grounds[['contact_id']]\nsample_submission_grounds['contact']=predictions_binary\nsample_submission_grounds.contact.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T17:01:54.037822Z","iopub.execute_input":"2023-03-03T17:01:54.038361Z","iopub.status.idle":"2023-03-03T17:01:54.056056Z","shell.execute_reply.started":"2023-03-03T17:01:54.038317Z","shell.execute_reply":"2023-03-03T17:01:54.054188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FINAL\nOn the final step, we merge two-player and player-ground predictions together.","metadata":{}},{"cell_type":"code","source":"submission = pd.concat([sample_submission_grounds, sample_submission], ignore_index=False)\nsubmission.sort_values(by='contact_id', inplace=True)\nsubmission.to_csv('Submission.csv')\nsubmission.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-03-03T17:05:05.760065Z","iopub.execute_input":"2023-03-03T17:05:05.760638Z","iopub.status.idle":"2023-03-03T17:05:05.953372Z","shell.execute_reply.started":"2023-03-03T17:05:05.760588Z","shell.execute_reply":"2023-03-03T17:05:05.951773Z"},"trusted":true},"execution_count":null,"outputs":[]}]}