{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-30T18:00:25.931477Z","iopub.execute_input":"2023-01-30T18:00:25.933141Z","iopub.status.idle":"2023-01-30T18:00:25.961822Z","shell.execute_reply.started":"2023-01-30T18:00:25.933019Z","shell.execute_reply":"2023-01-30T18:00:25.960143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using merged train and tracking data , apply a bernoulli naive bayes algorithm.**","metadata":{}},{"cell_type":"code","source":"# ==============================\n# read necessary libraries\n# ==============================\n\nimport os\nimport gc\nimport subprocess\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom IPython.display import Video, display\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import (\n    roc_auc_score,\n    matthews_corrcoef,\n)\n\nimport seaborn as sns\n%matplotlib inline\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:25.970103Z","iopub.execute_input":"2023-01-30T18:00:25.970603Z","iopub.status.idle":"2023-01-30T18:00:26.619119Z","shell.execute_reply.started":"2023-01-30T18:00:25.970563Z","shell.execute_reply":"2023-01-30T18:00:26.617469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# read data\n# ==============================\n\nTEhelmets = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv')\nTRhelmets = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv')\nsub = pd.read_csv('/kaggle/input/nfl-player-contact-detection/sample_submission.csv')\nTRtracking = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv')\nTEtracking = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv')\nTRvideoMeta = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_video_metadata.csv')\nTEvideoMeta = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_video_metadata.csv')\ntrainlabels = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:26.621049Z","iopub.execute_input":"2023-01-30T18:00:26.621454Z","iopub.status.idle":"2023-01-30T18:00:42.367294Z","shell.execute_reply.started":"2023-01-30T18:00:26.621400Z","shell.execute_reply":"2023-01-30T18:00:42.366002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1. Submission Dataset**","metadata":{}},{"cell_type":"code","source":"# =========================================\n# A characterisation of the Submission data\n# =========================================\n\nprint(sub.columns) \nprint('')\n\n# checking for missing values in content_df dataframe\nmissing_values = sub.isnull().sum()\n# Drop content_df rows with missing values\nsub = sub.dropna()\nprint(missing_values)\nprint('')\n# Print the data types of the content dataframe\nprint(f'Submission DataFrame Data Types: \\n{sub.dtypes}')\nprint('')\nprint(sub.shape) # (49588, 2)\nprint('')\nprint(sub[:7]) \nprint('')\nprint(sub['contact_id'].value_counts())\nprint('')\nprint(sub['contact'].value_counts()) # 0    49588","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:42.370637Z","iopub.execute_input":"2023-01-30T18:00:42.371066Z","iopub.status.idle":"2023-01-30T18:00:42.437007Z","shell.execute_reply.started":"2023-01-30T18:00:42.371028Z","shell.execute_reply":"2023-01-30T18:00:42.435142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2. trainlabels Dataset**","metadata":{}},{"cell_type":"code","source":"# ==========================================\n# A characterisation of the trainlabels data\n# ==========================================\n\nprint(trainlabels.columns) \nprint('')\n\n# checking for missing values in content_df dataframe\nmissing_values = trainlabels.isnull().sum()\nprint(missing_values)\nprint('')\n# Drop content_df rows with missing values\ntrainlabels = trainlabels.dropna()\n# checking for missing values in content_df dataframe\nmissing_values = trainlabels.isnull().sum()\nprint(missing_values)\nprint('')\n# Print the data types of the content dataframe\nprint(f'trainlabels DataFrame Data Types: \\n{trainlabels.dtypes}')\nprint('')\nprint(trainlabels.shape) # (4721618, 7)\nprint('')\nprint(trainlabels[:7])\nprint('')\n#print(trainlabels.columns)\nprint(trainlabels['step'].value_counts()) # 173\nprint('')\nprint(trainlabels['game_play'].value_counts()) # 240\nprint('')\nprint(trainlabels['datetime'].value_counts()) # 18666\nprint('')\nprint(trainlabels['contact_id'].value_counts()) # 4721618\nprint('')\nprint(trainlabels['contact'].value_counts())\nprint('')\nprint(trainlabels['nfl_player_id_1'].value_counts()) # 1687\nprint('')\nprint(trainlabels['nfl_player_id_2'].value_counts()) # 1646","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:42.439080Z","iopub.execute_input":"2023-01-30T18:00:42.439542Z","iopub.status.idle":"2023-01-30T18:00:52.709306Z","shell.execute_reply.started":"2023-01-30T18:00:42.439502Z","shell.execute_reply":"2023-01-30T18:00:52.707392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using just countplot to get the bars in the same order as \n# value_counts() output:\nsns.countplot(data=trainlabels, x='contact', order=trainlabels.contact.value_counts().index)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:52.712477Z","iopub.execute_input":"2023-01-30T18:00:52.713570Z","iopub.status.idle":"2023-01-30T18:00:53.371450Z","shell.execute_reply.started":"2023-01-30T18:00:52.713523Z","shell.execute_reply":"2023-01-30T18:00:53.369986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now, merge DataFrames using the merge() \n# merge sub and trainlabels datasets to further characterise \n# the contact_id and contact columns\n\nM1 = sub.merge(trainlabels, on = 'contact_id', how='left')\nM1 # 49588 rows × 8 columns","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:53.373415Z","iopub.execute_input":"2023-01-30T18:00:53.373914Z","iopub.status.idle":"2023-01-30T18:00:58.839107Z","shell.execute_reply.started":"2023-01-30T18:00:53.373874Z","shell.execute_reply":"2023-01-30T18:00:58.837767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**3. TRtracking Dataset**","metadata":{}},{"cell_type":"code","source":"# =========================================\n# A characterisation of the TRtracking data\n# =========================================\n\nprint(TRtracking.columns) \nprint('')\n\n# Rename the content_df 'id' column to 'content_ids'\n#TRtracking = TRtracking.rename(columns={'id': 'content_ids'})\n\n# checking for missing values in content_df dataframe\nmissing_values = TRtracking.isnull().sum()\n# Drop content_df rows with missing values\nTRtracking = TRtracking.dropna()\nprint(missing_values)\nprint('')\n# Print the data types of the content dataframe\nprint(f'TRtracking DataFrame Data Types: \\n{TRtracking.dtypes}')\nprint('')\nprint(TRtracking.shape) # (1353053, 17)\nprint('')\nprint(TRtracking[:7])\nprint('')\nprint(TRtracking['game_play'].value_counts()) # 240\nprint('')\nprint(TRtracking['game_key'].value_counts()) # 149\nprint('')\nprint(TRtracking['play_id'].value_counts()) # 233\nprint('')\nprint(TRtracking['nfl_player_id'].value_counts()) # 1687\nprint('')\nprint(TRtracking['datetime'].value_counts()) # 61279\nprint('')\nprint(TRtracking['step'].value_counts()) # 1032\nprint('')\nprint(TRtracking['team'].value_counts())\nprint('')\nprint(TRtracking['position'].value_counts())\nprint('')\nprint(TRtracking['jersey_number'].value_counts()) # 99\n\nprint('')\nprint(TRtracking['x_position'].value_counts()) # 12295\nprint('')\nprint(TRtracking['y_position'].value_counts()) # 6893","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:00:58.840984Z","iopub.execute_input":"2023-01-30T18:00:58.842486Z","iopub.status.idle":"2023-01-30T18:01:00.100941Z","shell.execute_reply.started":"2023-01-30T18:00:58.842407Z","shell.execute_reply":"2023-01-30T18:01:00.099534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge trainlabels and TRtracking datasets\n# based on earlier merge of sub and trainlebels datasets.\n\nM2 = M1.merge(TRtracking, how='left')\nM2 # 1090936 rows × 22 columns","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:00.102922Z","iopub.execute_input":"2023-01-30T18:01:00.103344Z","iopub.status.idle":"2023-01-30T18:01:01.417052Z","shell.execute_reply.started":"2023-01-30T18:01:00.103305Z","shell.execute_reply":"2023-01-30T18:01:01.415416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select variables / columns to be used in further analysis.\n\nM2T = M2[['step','team','x_position', 'y_position', 'speed', 'distance','direction','orientation','sa', 'contact_y']]\nM2T[:7]","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:01.419760Z","iopub.execute_input":"2023-01-30T18:01:01.420348Z","iopub.status.idle":"2023-01-30T18:01:01.700017Z","shell.execute_reply.started":"2023-01-30T18:01:01.420296Z","shell.execute_reply":"2023-01-30T18:01:01.698691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using just countplot to get the bars in the same order as\n# value_counts() output too:\nsns.countplot(data=M2T, x='contact_y', order=M2T.contact_y.value_counts().index)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:01.701744Z","iopub.execute_input":"2023-01-30T18:01:01.702145Z","iopub.status.idle":"2023-01-30T18:01:01.982041Z","shell.execute_reply.started":"2023-01-30T18:01:01.702099Z","shell.execute_reply":"2023-01-30T18:01:01.980627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# change datatype of the 'team' column to make it more amenable\n# for analysis.\n\nM2T['team'] = pd.to_numeric(M2T['team'], errors='coerce')\nM2T.dtypes ","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:01.983439Z","iopub.execute_input":"2023-01-30T18:01:01.983828Z","iopub.status.idle":"2023-01-30T18:01:02.839707Z","shell.execute_reply.started":"2023-01-30T18:01:01.983793Z","shell.execute_reply":"2023-01-30T18:01:02.838265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================================\n# Apply a TPOT derived BernoulliNB model\n# =========================================\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import BernoulliNB\nfrom sklearn.impute import SimpleImputer\n\n# NOTE: Make sure that the outcome column is labeled 'target' in the data file\n#tpot_data = pd.read_csv('PATH/TO/DATA/FILE', sep='COLUMN_SEPARATOR', dtype=np.float64)\n#features = tpot_data.drop('target', axis=1)\n#training_features, testing_features, training_target, testing_target = \\\n #           train_test_split(features, tpot_data['target'], random_state=23)\n\ntpot_data = M2T #pd.read_csv('PATH/TO/DATA/FILE', sep='COLUMN_SEPARATOR', dtype=np.float64)\nfeatures = M2T.drop('contact_y', axis=1)\ntraining_features, testing_features, training_target, testing_target = \\\n            train_test_split(features, tpot_data['contact_y'], random_state=23)\n\nimputer = SimpleImputer(strategy=\"median\")\nimputer.fit(training_features)\ntraining_features = imputer.transform(training_features)\ntesting_features = imputer.transform(testing_features)\n\n# Average CV score on the training set was: 0.5115774929745273\nexported_pipeline = BernoulliNB(alpha=0.01, fit_prior=False)\n# Fix random state in exported estimator\nif hasattr(exported_pipeline, 'random_state'):\n    setattr(exported_pipeline, 'random_state', 23)\n\nexported_pipeline.fit(training_features, training_target)\nresults = exported_pipeline.predict(testing_features)\n\nresults1 = results.copy()\nresults1","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:02.843561Z","iopub.execute_input":"2023-01-30T18:01:02.843986Z","iopub.status.idle":"2023-01-30T18:01:05.234723Z","shell.execute_reply.started":"2023-01-30T18:01:02.843949Z","shell.execute_reply":"2023-01-30T18:01:05.233251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**4. TEtracking Dataset**","metadata":{}},{"cell_type":"code","source":"# =========================================\n# A characterisation of the TEtracking data\n# =========================================\n\nprint(TEtracking.columns) \nprint('')\n\n# Rename the content_df 'id' column to 'content_ids'\n#TRtracking = TRtracking.rename(columns={'id': 'content_ids'})\n\n# checking for missing values in content_df dataframe\nmissing_values = TEtracking.isnull().sum()\n# Drop content_df rows with missing values\nTEtracking = TEtracking.dropna()\nprint(missing_values)\nprint('')\n# Print the data types of the content dataframe\nprint(f'TEtracking DataFrame Data Types: \\n{TEtracking.dtypes}')\nprint('')\nprint(TEtracking.shape) # (14872, 17)\nprint(TEtracking[:7])\nprint('')\nprint(TEtracking['game_play'].value_counts()) # 2\nprint('')\nprint(TEtracking['game_key'].value_counts()) # 2\nprint('')\nprint(TEtracking['play_id'].value_counts()) # 2\nprint('')\nprint(TEtracking['nfl_player_id'].value_counts()) # \nprint('')\nprint(TEtracking['datetime'].value_counts()) # 676\nprint('')\nprint(TEtracking['step'].value_counts()) # 398\nprint('')\nprint(TEtracking['team'].value_counts())\nprint('')\nprint(TEtracking['position'].value_counts())\nprint('')\nprint(TEtracking['jersey_number'].value_counts()) #\n","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:05.236563Z","iopub.execute_input":"2023-01-30T18:01:05.238001Z","iopub.status.idle":"2023-01-30T18:01:05.328565Z","shell.execute_reply.started":"2023-01-30T18:01:05.237933Z","shell.execute_reply":"2023-01-30T18:01:05.327111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================\n# Prepare TEST data for use with model\n# ====================================\n\n\nTEST = TEtracking[['step','team','x_position', 'y_position', 'speed', 'distance','direction','orientation','sa']]\nTEST","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:05.330365Z","iopub.execute_input":"2023-01-30T18:01:05.331870Z","iopub.status.idle":"2023-01-30T18:01:05.365746Z","shell.execute_reply.started":"2023-01-30T18:01:05.331805Z","shell.execute_reply":"2023-01-30T18:01:05.364485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===========================\n# Prepare data for submission\n# ===========================\n\ntesting_features = TEST\n#results1\n\ndfAAAAA = pd.DataFrame(results1)\n#dfAAAAA\n\nSUBMISSION1 = pd.DataFrame()\nSUBMISSION1['contact_id']=sub['contact_id'] \nSUBMISSION1['contact']=dfAAAAA[0] #results1\nSUBMISSION1","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:05.367399Z","iopub.execute_input":"2023-01-30T18:01:05.367932Z","iopub.status.idle":"2023-01-30T18:01:05.395081Z","shell.execute_reply.started":"2023-01-30T18:01:05.367891Z","shell.execute_reply":"2023-01-30T18:01:05.393636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===========================================\n# A distribution of contacts and non-contacts\n# ===========================================\n\n# by number of occurences\nprint(SUBMISSION1['contact'].value_counts(0))\nprint('')\n# by % of total number of occurences\nprint(SUBMISSION1['contact'].value_counts(1))\n\n# Using just countplot to get the bars in the same order as \n#.value_counts() output:\nsns.countplot(data=SUBMISSION1, x='contact', order=SUBMISSION1.contact.value_counts().index)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:05.396217Z","iopub.execute_input":"2023-01-30T18:01:05.396576Z","iopub.status.idle":"2023-01-30T18:01:05.578334Z","shell.execute_reply.started":"2023-01-30T18:01:05.396544Z","shell.execute_reply":"2023-01-30T18:01:05.577100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#SUBMISSION1.to_csv('C:/Users/Owner/Desktop/NFL - PLAYER CONTACT DETECTION/submissionClassifier1.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T18:01:05.581074Z","iopub.execute_input":"2023-01-30T18:01:05.581482Z","iopub.status.idle":"2023-01-30T18:01:05.586209Z","shell.execute_reply.started":"2023-01-30T18:01:05.581446Z","shell.execute_reply":"2023-01-30T18:01:05.585157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}