{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.metrics import matthews_corrcoef\nimport torch, torchvision\nimport torch.nn as nn\nimport xgboost as xgb\nfrom xgboost import XGBRegressor\nfrom sklearn import preprocessing\nfrom torch import Tensor\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom torch.utils.data.dataloader import default_collate\nfrom sklearn import preprocessing\nimport gc","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:23:28.440027Z","iopub.execute_input":"2023-02-06T16:23:28.441032Z","iopub.status.idle":"2023-02-06T16:23:31.600891Z","shell.execute_reply.started":"2023-02-06T16:23:28.440951Z","shell.execute_reply":"2023-02-06T16:23:31.599979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_kaggle = \"/kaggle/input/nfl-player-contact-detection/\"\ndata = ''\ntrain_tracking = pd.read_csv(f\"{data_kaggle}train_player_tracking.csv\")\ntrain_helmets = pd.read_csv(f\"{data_kaggle}train_baseline_helmets.csv\")\ntrain_labels = pd.read_csv(f\"{data_kaggle}train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:23:31.602467Z","iopub.execute_input":"2023-02-06T16:23:31.603568Z","iopub.status.idle":"2023-02-06T16:23:56.223297Z","shell.execute_reply.started":"2023-02-06T16:23:31.603533Z","shell.execute_reply":"2023-02-06T16:23:56.222066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UTILS","metadata":{}},{"cell_type":"code","source":"def joindfs(tracking, helmets, labels, training=True):\n    fps = 59.94\n    frame_delta = 6\n\n    tracking = tracking.copy()\n    helmets = helmets.copy()\n    labels = labels.copy()\n\n    df = labels[['contact_id', 'contact']]\n\n    # taking the values required from the label df and converting steps to frames\n    df = df.copy()\n    df[\"game_play\"] = df.contact_id.apply(lambda x: \"_\".join(x.split(\"_\")[0:2]))\n    df[\"step\"] = df.contact_id.apply(lambda x: x.split(\"_\")[2]).astype(int)\n    df[\"nfl_player_id_1\"] = df.contact_id.apply(lambda x: x.split(\"_\")[3]).astype(int)\n    df[\"nfl_player_id_2\"] = df.contact_id.apply(lambda x: x.split(\"_\")[4])\n    df[\"nfl_player_id_2\"] = df[\"nfl_player_id_2\"].apply(lambda x: -1 if x == \"G\" else x).astype(int)\n    \n\n    # Columns to take from the tracking df, you can add speed and acceleration if you want but remember to rename the columns below\n    tracking_columns = [\"nfl_player_id\", \"x_position\", \"y_position\", \"game_play\", \"step\", 'team', 'orientation', 'direction', 'acceleration', 'sa', 'speed', 'distance']\n\n    # tracking df for each player\n    df_tracking_1 = tracking[tracking_columns]\n    df_tracking_1 = df_tracking_1.rename(columns={\"nfl_player_id\": \"nfl_player_id_1\", \"x_position\": \"x_position_1\", \"y_position\": \"y_position_1\", 'team': 'team_1', 'orientation': 'orientation_1', 'direction': 'direction_1', 'acceleration': 'acceleration_1', 'sa': 'sa_1', 'speed':'speed_1', 'distance': 'distance_1'})\n    df_tracking_2 = tracking[tracking_columns]\n    df_tracking_2 = df_tracking_2.rename(columns={\"nfl_player_id\": \"nfl_player_id_2\", \"x_position\": \"x_position_2\", \"y_position\": \"y_position_2\", 'team': 'team_2', 'orientation': 'orientation_2', 'direction': 'direction_2', 'acceleration': 'acceleration_2', 'sa': 'sa_2', 'speed':'speed_2', 'distance': 'distance_2'})\n\n    # merge the label df with the tracking ones for each player\n    df = pd.merge(df_tracking_1, df, how=\"right\", on=['nfl_player_id_1', 'game_play', 'step'])\n    df = pd.merge(df_tracking_2, df, how=\"right\", on=['nfl_player_id_2', 'game_play', 'step'])\n    \n    df[\"frame\"] = df[\"step\"].apply(lambda x: 300 + int(x * 0.1 * fps / frame_delta) * frame_delta)\n    \n    df['same_team'] = np.where(df['team_1'] == df['team_2'], 1, -1)\n    df['rel_orientation'] = df['orientation_1'] - df['orientation_2']\n    df['rel_direction'] = df['direction_1'] - df['direction_2']\n\n    # calculate the player to player distance\n    df[\"distance\"] = ((df.x_position_2 - df.x_position_1) ** 2 + (df.y_position_2 - df.y_position_1) ** 2) ** 0.5\n\n    helmets_columns = ['game_play', 'view', 'frame', 'nfl_player_id', 'left', 'width', 'top', 'height']\n    helmets = helmets.astype({'left': 'int32', 'width': 'int32', 'top': 'int32', 'height': 'int32'})\n    \n    df_helmets_1 = helmets[helmets_columns]\n    df_helmets_1 = df_helmets_1.rename(columns={\"nfl_player_id\": \"nfl_player_id_1\", \"left\": \"left_1\", \"width\": \"width_1\", \"top\": \"top_1\", \"height\": \"height_1\"})\n    df_helmets_2 = helmets[helmets_columns]\n    df_helmets_2 = df_helmets_2.rename(columns={\"nfl_player_id\": \"nfl_player_id_2\", \"left\": \"left_2\", \"width\": \"width_2\", \"top\": \"top_2\", \"height\": \"height_2\"})\n\n    df = pd.merge(df_helmets_1, df, how=\"right\", on=['nfl_player_id_1', 'game_play', 'frame'])\n    df = pd.merge(df_helmets_2, df, how=\"right\", on=['nfl_player_id_2', 'game_play', 'frame', 'view'])\n    df['view'] = preprocessing.LabelEncoder().fit_transform(df['view'])\n    \n    #df = df.groupby('contact_id', as_index=False).mean()\n    #df[\"game_play\"] = df.contact_id.apply(lambda x: \"_\".join(x.split(\"_\")[0:2]))\n    #df = pd.get_dummies(df, columns=['view'])\n    # uncomment this if you are dealing with both at the same time\n    return df\n\n    df_player = df[df.nfl_player_id_2 != -1]\n    \n    df_ground = df[df.nfl_player_id_2 == -1]\n    df_ground = df_ground.drop(labels=['left_2', 'width_2', 'top_2', 'height_2', 'x_position_2', 'y_position_2', 'distance', 'distance_2'], axis=1)\n    \n    \n\n    return df_player, df_ground","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:23:56.224885Z","iopub.execute_input":"2023-02-06T16:23:56.225199Z","iopub.status.idle":"2023-02-06T16:23:56.244025Z","shell.execute_reply.started":"2023-02-06T16:23:56.225171Z","shell.execute_reply":"2023-02-06T16:23:56.242969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_contacts(data_frame, n_samples_per_contact):\n    # Calculate the total number of contacts in the DataFrame\n    total_contacts = data_frame['contact'].sum()\n    \n    # Extract rows with contact = 1 (contacts) and contact = 0 (non-contacts)\n    contacts = data_frame[data_frame['contact'] == 1]\n    non_contacts = data_frame[data_frame['contact'] == 0]\n    \n    # Sample n_samples_per_contact times the number of non-contacts to create a balanced dataset\n    sampled_non_contacts = non_contacts.sample(n_samples_per_contact * total_contacts)\n    \n    # Concatenate the sampled contacts and non-contacts to create the final DataFrame\n    sampled_data_frame = pd.concat([contacts, sampled_non_contacts])\n    \n    return sampled_data_frame","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:23:56.246536Z","iopub.execute_input":"2023-02-06T16:23:56.246971Z","iopub.status.idle":"2023-02-06T16:23:56.265395Z","shell.execute_reply.started":"2023-02-06T16:23:56.246940Z","shell.execute_reply":"2023-02-06T16:23:56.264215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_test_split(df, feature_column, n=20, test_percent=0.2):\n    # Initialize empty DataFrames for training and testing sets\n    train, test = pd.DataFrame(), pd.DataFrame()\n\n    # Get the unique game plays from the DataFrame\n    plays = df['game_play'].unique()\n    length_plays = len(plays)\n    print(length_plays)\n\n    # Determine the index at which to split the data into training and testing sets\n    train_split = length_plays - int(length_plays * test_percent)\n\n    # Populate the training set with data for plays before the split\n    for play in plays[:train_split]:\n        df_play = df[df['game_play'] == play]\n        train = pd.concat([train, df_play])\n\n    # Populate the testing set with data for plays after the split\n    for play in plays[train_split:length_plays]:\n        df_play = df[df['game_play'] == play]\n        test = pd.concat([test, df_play])\n\n    # If a testing set is required (test_percent > 0), sample contacts in the training set\n    if test_percent > 0:\n        train = sample_contacts(train, n)\n        return train[feature_column], train['contact'], test[feature_column], test['contact']\n    else:\n        # If no testing set is required, return the training set as is\n        return train, train['contact']","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:23:56.266639Z","iopub.execute_input":"2023-02-06T16:23:56.267112Z","iopub.status.idle":"2023-02-06T16:23:56.286537Z","shell.execute_reply.started":"2023-02-06T16:23:56.267074Z","shell.execute_reply":"2023-02-06T16:23:56.284888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Joining three DataFrames: train_tracking, train_helmets, and train_labels\n# to create a single DataFrame named 'df'.\ndf = joindfs(train_tracking, train_helmets, train_labels)\n\n# List of feature column names that will be used for training the model\nFEATURES = ['distance', 'frame', 'step', 'left_1', 'width_1', 'top_1', 'height_1', \n            'step', 'left_2', 'width_2', 'top_2', 'height_2', 'view', 'same_team', \n            'rel_orientation', 'orientation_1', 'orientation_2', 'rel_direction', \n            'direction_1', 'direction_2', 'speed_1', 'speed_2', 'acceleration_1', \n            'acceleration_2', 'sa_1', 'sa_2', 'x_position_1', 'y_position_1', \n            'x_position_2', 'y_position_2', 'distance_1', 'distance_2']\n\n# Split the 'df' DataFrame into training and testing sets with the specified feature columns.\n# 'X_df' contains the feature columns, and 'y_train' contains the target variable.\nX_df, y_train = train_test_split(df, FEATURES, test_percent=0)\n\n# Extract the feature columns from 'X_df' and fill any missing values with 0.\nX_train = X_df[FEATURES]\nX_train.fillna(0, inplace=True)\n\n# Perform garbage collection to free up memory.\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:23:56.288499Z","iopub.execute_input":"2023-02-06T16:23:56.288952Z","iopub.status.idle":"2023-02-06T16:24:42.783490Z","shell.execute_reply.started":"2023-02-06T16:23:56.288917Z","shell.execute_reply":"2023-02-06T16:24:42.781645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PLAYER MODEL","metadata":{}},{"cell_type":"code","source":"# Create a list 'cols' containing column names that end with '_1' but are not 'nfl_player_id_1'.\ncols = [i[:-2] for i in X_train.columns if i[-2:] == '_1' and i != 'nfl_player_id_1']\n\n# Create a new DataFrame 'new_x_train' by taking the absolute difference between corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_diff' suffix.\nnew_x_train = pd.DataFrame(np.abs(X_train[[i + '_1' for i in cols]].values - X_train[[i + '_2' for i in cols]].values),\n                          columns=[i + '_diff' for i in cols])\n\n# Add the new '_diff' columns to the 'X_train' DataFrame.\nX_train[[i + '_diff' for i in cols]] = new_x_train\n\n# Create a list 'cols' containing column names without the '_1' or '_2' suffix.\ncols = ['x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration', 'sa']\n\n# Create a new DataFrame 'new_x_train' by taking the element-wise product of corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_prod' suffix.\nnew_x_train = pd.DataFrame(X_train[[i + '_1' for i in cols]].values * X_train[[i + '_2' for i in cols]].values,\n                          columns=[i + '_prod' for i in cols])\n\n# Add the new '_prod' columns to the 'X_train' DataFrame.\nX_train[[i + '_prod' for i in cols]] = new_x_train\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:27:58.017160Z","iopub.execute_input":"2023-02-06T16:27:58.017635Z","iopub.status.idle":"2023-02-06T16:27:59.958827Z","shell.execute_reply.started":"2023-02-06T16:27:58.017603Z","shell.execute_reply":"2023-02-06T16:27:59.957680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an instance of the XGBoost classifier\nCLF = xgb.XGBClassifier()","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:28:01.523111Z","iopub.execute_input":"2023-02-06T16:28:01.523551Z","iopub.status.idle":"2023-02-06T16:28:01.530149Z","shell.execute_reply.started":"2023-02-06T16:28:01.523513Z","shell.execute_reply":"2023-02-06T16:28:01.528292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the XGBoost classifier (CLF) to the training data\nCLF.fit(X_train.values,  # Features of the training data\n         y_train.values,  # Labels of the training data\n         eval_set=[(X_train.values, y_train.values)],  # Evaluation set for monitoring model performance\n         verbose=1)  # Verbosity level (1 for displaying training progress)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T16:28:01.531812Z","iopub.execute_input":"2023-02-06T16:28:01.532212Z","iopub.status.idle":"2023-02-06T17:02:28.211262Z","shell.execute_reply.started":"2023-02-06T16:28:01.532176Z","shell.execute_reply":"2023-02-06T17:02:28.210425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(CLF.feature_importances_)\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:28.212262Z","iopub.execute_input":"2023-02-06T17:02:28.213341Z","iopub.status.idle":"2023-02-06T17:02:28.226665Z","shell.execute_reply.started":"2023-02-06T17:02:28.213282Z","shell.execute_reply":"2023-02-06T17:02:28.225872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = CLF.predict(X_train.values)\nprint(matthews_corrcoef(y_train, test_pred))","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:28.425633Z","iopub.execute_input":"2023-02-06T17:02:28.426136Z","iopub.status.idle":"2023-02-06T17:02:44.488757Z","shell.execute_reply.started":"2023-02-06T17:02:28.426100Z","shell.execute_reply":"2023-02-06T17:02:44.486581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tracking = pd.read_csv(f\"{data_kaggle}test_player_tracking.csv\")\ntest_helmets = pd.read_csv(f\"{data_kaggle}test_baseline_helmets.csv\")\ntest_labels = pd.read_csv(f\"{data_kaggle}sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:44.490873Z","iopub.execute_input":"2023-02-06T17:02:44.491287Z","iopub.status.idle":"2023-02-06T17:02:44.712106Z","shell.execute_reply.started":"2023-02-06T17:02:44.491250Z","shell.execute_reply":"2023-02-06T17:02:44.711022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = joindfs(test_tracking, test_helmets, test_labels)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:44.713434Z","iopub.execute_input":"2023-02-06T17:02:44.714196Z","iopub.status.idle":"2023-02-06T17:02:45.178097Z","shell.execute_reply.started":"2023-02-06T17:02:44.714136Z","shell.execute_reply":"2023-02-06T17:02:45.177070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_df, y_test = train_test_split(test_df, FEATURES, test_percent=0)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.179408Z","iopub.execute_input":"2023-02-06T17:02:45.179787Z","iopub.status.idle":"2023-02-06T17:02:45.247355Z","shell.execute_reply.started":"2023-02-06T17:02:45.179755Z","shell.execute_reply":"2023-02-06T17:02:45.246152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = X_test_df[FEATURES]","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.249027Z","iopub.execute_input":"2023-02-06T17:02:45.249331Z","iopub.status.idle":"2023-02-06T17:02:45.271629Z","shell.execute_reply.started":"2023-02-06T17:02:45.249305Z","shell.execute_reply":"2023-02-06T17:02:45.269627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.273791Z","iopub.execute_input":"2023-02-06T17:02:45.274263Z","iopub.status.idle":"2023-02-06T17:02:45.285914Z","shell.execute_reply.started":"2023-02-06T17:02:45.274224Z","shell.execute_reply":"2023-02-06T17:02:45.284873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list 'cols' containing column names that end with '_1' but are not 'nfl_player_id_1'.\ncols = [i[:-2] for i in X_test.columns if i[-2:] == '_1' and i != 'nfl_player_id_1']\n\n# Create a new DataFrame 'new_x_train' by taking the absolute difference between corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_diff' suffix.\nnew_x_test = pd.DataFrame(np.abs(X_test[[i + '_1' for i in cols]].values - X_test[[i + '_2' for i in cols]].values),\n                          columns=[i + '_diff' for i in cols])\n\n# Add the new '_diff' columns to the 'X_test' DataFrame.\nX_test[[i + '_diff' for i in cols]] = new_x_test\n\n# Create a list 'cols' containing column names without the '_1' or '_2' suffix.\ncols = ['x_position', 'y_position', 'speed', 'direction', 'orientation', 'acceleration', 'sa']\n\n# Create a new DataFrame 'new_x_train' by taking the element-wise product of corresponding '_1' and '_2' columns.\n# The resulting columns are named with '_prod' suffix.\nnew_x_test = pd.DataFrame(X_test[[i + '_1' for i in cols]].values * X_test[[i + '_2' for i in cols]].values,\n                          columns=[i + '_prod' for i in cols])\n\n# Add the new '_prod' columns to the 'X_test' DataFrame.\nX_test[[i + '_prod' for i in cols]] = new_x_test\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.292296Z","iopub.execute_input":"2023-02-06T17:02:45.292735Z","iopub.status.idle":"2023-02-06T17:02:45.366641Z","shell.execute_reply.started":"2023-02-06T17:02:45.292677Z","shell.execute_reply":"2023-02-06T17:02:45.365007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = CLF.predict(X_test.values)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.368324Z","iopub.execute_input":"2023-02-06T17:02:45.368644Z","iopub.status.idle":"2023-02-06T17:02:45.539072Z","shell.execute_reply.started":"2023-02-06T17:02:45.368617Z","shell.execute_reply":"2023-02-06T17:02:45.537672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test = pd.DataFrame()\nsample_test['contact_id'] = X_test_df['contact_id']\nsample_test['contact'] = predictions ","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.541029Z","iopub.execute_input":"2023-02-06T17:02:45.541397Z","iopub.status.idle":"2023-02-06T17:02:45.561530Z","shell.execute_reply.started":"2023-02-06T17:02:45.541367Z","shell.execute_reply":"2023-02-06T17:02:45.559233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test = sample_test.groupby('contact_id', as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.564225Z","iopub.execute_input":"2023-02-06T17:02:45.565262Z","iopub.status.idle":"2023-02-06T17:02:45.634368Z","shell.execute_reply.started":"2023-02-06T17:02:45.565200Z","shell.execute_reply":"2023-02-06T17:02:45.633349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels.set_index('contact_id', inplace=True)\nsample_test.set_index('contact_id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.635657Z","iopub.execute_input":"2023-02-06T17:02:45.635969Z","iopub.status.idle":"2023-02-06T17:02:45.645215Z","shell.execute_reply.started":"2023-02-06T17:02:45.635945Z","shell.execute_reply":"2023-02-06T17:02:45.642608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test = sample_test.reindex(test_labels.index)\nsample_test.reset_index(inplace=True)\nsample_test['contact'] = (sample_test.contact > 0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.649326Z","iopub.execute_input":"2023-02-06T17:02:45.650310Z","iopub.status.idle":"2023-02-06T17:02:45.694933Z","shell.execute_reply.started":"2023-02-06T17:02:45.650241Z","shell.execute_reply":"2023-02-06T17:02:45.693272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T17:02:45.697895Z","iopub.execute_input":"2023-02-06T17:02:45.698424Z","iopub.status.idle":"2023-02-06T17:02:45.774121Z","shell.execute_reply.started":"2023-02-06T17:02:45.698372Z","shell.execute_reply":"2023-02-06T17:02:45.772530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}