{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-04T09:13:40.118190Z","iopub.execute_input":"2023-02-04T09:13:40.118757Z","iopub.status.idle":"2023-02-04T09:13:40.366889Z","shell.execute_reply.started":"2023-02-04T09:13:40.118630Z","shell.execute_reply":"2023-02-04T09:13:40.365412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# read Libraries\n# ==============================\n\nimport os\nimport gc\nimport subprocess\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom IPython.display import Video, display\n\nfrom scipy.optimize import minimize\n# import cv2\nfrom glob import glob\n#from tqdm import tqdm\n\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import (\n    roc_auc_score,\n    matthews_corrcoef,\n)\n'''\n! pip install xgboost\nimport xgboost as xgb\n\n! pip install torch\nimport torch\n\nif torch.cuda.is_available():\n    import cupy \n    import cudf\n    from cuml import ForestInference\n'''    \n\nimport seaborn as sns\n%matplotlib inline\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:13:40.369347Z","iopub.execute_input":"2023-02-04T09:13:40.369715Z","iopub.status.idle":"2023-02-04T09:13:41.952710Z","shell.execute_reply.started":"2023-02-04T09:13:40.369681Z","shell.execute_reply":"2023-02-04T09:13:41.951707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# read data\n# ==============================\n\nTEhelmets = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv')\nTRhelmets = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv')\nsub = pd.read_csv('/kaggle/input/nfl-player-contact-detection/sample_submission.csv')\nTRtracking = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv')\nTEtracking = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv')\nTRvideoMeta = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_video_metadata.csv')\nTEvideoMeta = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_video_metadata.csv')\ntrainlabels = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_labels.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:13:41.954556Z","iopub.execute_input":"2023-02-04T09:13:41.955901Z","iopub.status.idle":"2023-02-04T09:14:10.952979Z","shell.execute_reply.started":"2023-02-04T09:13:41.955841Z","shell.execute_reply":"2023-02-04T09:14:10.951708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TRtracking Dataset**","metadata":{}},{"cell_type":"code","source":"# ================================\n# preprocess of TRtracking dataset\n# ================================\n\n# view columns of TRtracking dataset \nprint(TRtracking.columns) \nprint('')\n\n# checking for missing values in content_df dataframe\nmissing_values = TRtracking.isnull().sum()\n\n# Drop content_df rows with missing values\nTRtracking = TRtracking.dropna()\n\nprint(missing_values)\nprint('')\n# Print the data types of the content dataframe\nprint(f'TRtracking DataFrame Data Types: \\n{TRtracking.dtypes}')\nprint('')\nprint(TRtracking.shape)\n\n# view first 7 rows of TRtracking dataframe\nTRtracking[:7] ","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:14:10.956110Z","iopub.execute_input":"2023-02-04T09:14:10.956517Z","iopub.status.idle":"2023-02-04T09:14:11.838146Z","shell.execute_reply.started":"2023-02-04T09:14:10.956479Z","shell.execute_reply":"2023-02-04T09:14:11.836734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TRtracking Dataset - team==home**","metadata":{}},{"cell_type":"code","source":"# ===========================================\n# preprocess of TRtracking team==home dataset\n# ===========================================\n\n'''team==home'''\n# derive the team==home dataset\nTRtrackinghome = TRtracking.loc[TRtracking['team']=='home']# 676599 rows × 17 columns\n\n# derive the 7 top most values of the 'step' variable of the team==home dataset\ndfTRh1 = TRtrackinghome['step'].value_counts()[:7]\n\n# select rows where the value of the 'step' variable is among the \n# top most 7 for both team==home and train labels datasets.\ndfTRh17 = TRtrackinghome.loc[TRtrackinghome['step'].isin([14, 9, 19, 18, 17, 16, 15])]\ndfTRh17trainlabels = trainlabels.loc[trainlabels['step'].isin([14, 9, 19, 18, 17, 16, 15])]\n\ngc.collect()\n\n# merge the two datasets\ndfTRh1M = dfTRh17.merge(dfTRh17trainlabels,on='game_play',how='inner')\n\n# select variables of the merged dataset for further analysis\ndfTRh1M_Data = dfTRh1M[['contact_id','step_x','x_position',\n       'y_position', 'speed', 'distance', 'direction', 'orientation',\n       'acceleration', 'sa','contact']]\n\n# what is the correlation between the variables\nDATA_TRH = dfTRh1M_Data[['step_x','speed','distance','direction','acceleration','sa','contact']]#.copy()\nprint(DATA_TRH.corr())\n\nheatmap1 = sns.heatmap(DATA_TRH.corr(),vmax=1, annot=True)\nheatmap1.set_title('Correlation Heatmap - TRtracking team==home ', fontdict={'fontsize':12}, pad=12)\n\n# what is the correlation between the variables\nDATA_TRH = dfTRh1M_Data[['step_x','speed','distance','direction','acceleration','sa','contact']]#.copy()\nprint(DATA_TRH.corr())\n\nheatmap1 = sns.heatmap(DATA_TRH.corr(),vmax=1, annot=True)\nheatmap1.set_title('Correlation Heatmap - TRtracking team==home ', fontdict={'fontsize':12}, pad=12)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:14:11.840025Z","iopub.execute_input":"2023-02-04T09:14:11.840531Z","iopub.status.idle":"2023-02-04T09:15:20.839288Z","shell.execute_reply.started":"2023-02-04T09:14:11.840482Z","shell.execute_reply":"2023-02-04T09:15:20.837897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TRtracking Dataset - team==home - logistic regression model**","metadata":{}},{"cell_type":"code","source":"# =========================================================\n# Logistic Regression Model - TRtracking team==home dataset\n# =========================================================\n\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, f1_score\n\n\n# Select relevant features\nX = DATA_TRH[['step_x','speed', 'distance','direction','acceleration','sa']] #[['language_x', 'category']]]]\ny = DATA_TRH['contact']\n\n# Split the data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Train the model\nmodel = LogisticRegression()\nmodel.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Calculate evaluation metrics\naccuracy = accuracy_score(y_test, y_pred)\nf1 = f1_score(y_test, y_pred, average='weighted')\n\n# Print the evaluation metrics on the test set\nprint('Accuracy:', accuracy)\nprint('F1 Score:', f1)\nprint('')\nprint(y_pred)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:15:20.841355Z","iopub.execute_input":"2023-02-04T09:15:20.842529Z","iopub.status.idle":"2023-02-04T09:18:08.005221Z","shell.execute_reply.started":"2023-02-04T09:15:20.842478Z","shell.execute_reply":"2023-02-04T09:18:08.003761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TEtracking Dataset**","metadata":{}},{"cell_type":"code","source":"# ===========================================\n# preprocess of TEtracking team==home dataset\n# ===========================================\n\nprint(TEtracking.columns) \nprint('')\n\n# checking for missing values in content_df dataframe\nmissing_values = TEtracking.isnull().sum()\n# Drop content_df rows with missing values\nTEtracking = TEtracking.dropna()\nprint(missing_values)\nprint('')\n# Print the data types of the content dataframe\nprint(f'TEtracking DataFrame Data Types: \\n{TEtracking.dtypes}')\nprint('')\nprint(TEtracking.shape)\nTEtracking[:7] #.head(2)\n\n'''team==home'''\n# derive the team==home dataset\nTEtrackinghome = TEtracking.loc[TEtracking['team']=='home']# 676599 rows × 17 columns\n\n# derive the 7 top most values of the 'step' variable of the team==home dataset\ndfTEh1 = TEtrackinghome['step'].value_counts()[:7]\n\n# select rows where the value of the 'step' variable is among the \n# top most 7 for both team==home and train labels datasets.\ndfTEh17 = TEtrackinghome.loc[TEtrackinghome['step'].isin([14, 9, 19, 18, 17, 16, 15])]\ndfTEh17trainlabels = trainlabels.loc[trainlabels['step'].isin([14, 9, 19, 18, 17, 16, 15])]\n\ngc.collect()\n\n# merge the two datasets\ndfTEh1M = dfTEh17.merge(dfTEh17trainlabels,on='game_play',how='inner')\n\n# select variables of the merged dataset for further analysis\ndfTEh1M_Data = dfTEh1M[['contact_id','step_x','x_position',\n       'y_position', 'speed', 'distance', 'direction', 'orientation',\n       'acceleration', 'sa','contact']]\n\n# what is the correlation between the variables\nDATA_TEH = dfTEh1M_Data[['step_x','speed','distance','direction','acceleration','sa']]#.copy()\nprint(DATA_TEH.corr())\n\nheatmap2 = sns.heatmap(DATA_TEH.corr(),vmax=1, annot=True)\nheatmap2.set_title('Correlation Heatmap - TEtracking team==home ', fontdict={'fontsize':12}, pad=12)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:18:08.007200Z","iopub.execute_input":"2023-02-04T09:18:08.008170Z","iopub.status.idle":"2023-02-04T09:18:09.286498Z","shell.execute_reply.started":"2023-02-04T09:18:08.008117Z","shell.execute_reply":"2023-02-04T09:18:09.285577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the structure of the two datasets\ndfTEh1M_Data.shape, DATA_TEH.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:18:09.288088Z","iopub.execute_input":"2023-02-04T09:18:09.288966Z","iopub.status.idle":"2023-02-04T09:18:09.296548Z","shell.execute_reply.started":"2023-02-04T09:18:09.288924Z","shell.execute_reply":"2023-02-04T09:18:09.295061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Competition Objective \nFor each contact_id in the test set, you must predict a contact (0 = no contact between players, 1 = contact between the two players for that contact_id. The file should contain a header","metadata":{}},{"cell_type":"code","source":"# substitute the DATA_TEH dataset for the X_test dataset \nX_test = DATA_TEH\n\n# derive the required predictions for submission\npredictionsHS = model.predict(X_test)\npredictionsHS","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:18:09.298973Z","iopub.execute_input":"2023-02-04T09:18:09.299902Z","iopub.status.idle":"2023-02-04T09:18:09.341505Z","shell.execute_reply.started":"2023-02-04T09:18:09.299840Z","shell.execute_reply":"2023-02-04T09:18:09.340207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a submission dataframe using the 2 column headings as required by \n# the competition rules. \nSUBMISSION = pd.DataFrame()\nSUBMISSION['contact_id']=dfTEh1M_Data['contact_id'] \nSUBMISSION['contact']=predictionsHS\n\n# view the 7 first rows of the dataframe\nprint(SUBMISSION[:7])\n\n# save the dataframe for submission\nSUBMISSION.to_csv('submission.csv', index=False)      ","metadata":{"execution":{"iopub.status.busy":"2023-02-04T09:18:09.345234Z","iopub.execute_input":"2023-02-04T09:18:09.346957Z","iopub.status.idle":"2023-02-04T09:18:09.827807Z","shell.execute_reply.started":"2023-02-04T09:18:09.346901Z","shell.execute_reply":"2023-02-04T09:18:09.826688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}