{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2023-02-22T01:54:04.520905Z","iopub.execute_input":"2023-02-22T01:54:04.521468Z","iopub.status.idle":"2023-02-22T01:54:04.544921Z","shell.execute_reply.started":"2023-02-22T01:54:04.521376Z","shell.execute_reply":"2023-02-22T01:54:04.544070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:04.546509Z","iopub.execute_input":"2023-02-22T01:54:04.546829Z","iopub.status.idle":"2023-02-22T01:54:04.558694Z","shell.execute_reply.started":"2023-02-22T01:54:04.546799Z","shell.execute_reply":"2023-02-22T01:54:04.557751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**READ THE DATA**","metadata":{}},{"cell_type":"code","source":"TRtracking = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv')\nTEtracking = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv')\nTRhelmets = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv')\nTEhelmets = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv')\nTRvideoMeta = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_video_metadata.csv')\nTEvideoMeta = pd.read_csv('/kaggle/input/nfl-player-contact-detection/test_video_metadata.csv')\nsub = pd.read_csv('/kaggle/input/nfl-player-contact-detection/sample_submission.csv')\ntrainlabels = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:04.559861Z","iopub.execute_input":"2023-02-22T01:54:04.560252Z","iopub.status.idle":"2023-02-22T01:54:17.644539Z","shell.execute_reply.started":"2023-02-22T01:54:04.560212Z","shell.execute_reply":"2023-02-22T01:54:17.643335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SELECT VARIABLES OF INTEREST FROM ALL DATASETS**\n**PREPROCESS THE SELECTED DATA**","metadata":{}},{"cell_type":"markdown","source":"**PREPROCESS THE SELECTED DATA**","metadata":{}},{"cell_type":"markdown","source":"**TRAIN DATASETS**","metadata":{}},{"cell_type":"code","source":"### SELECT VARIABLES - Training Datasets\nTRtrackingS = TRtracking[['game_play','step','team','position','x_position',\n                          'y_position']] # 'play_id',\nTRhelmetsS = TRhelmets[['game_play','view','left', 'width', 'top', 'height']]\nTRvideoMetaS = TRvideoMeta[['game_play','view','start_time', 'end_time','snap_time']]\n\n### MERGE HELMET AND VIDEO DATA\ndftr1 = pd.merge(TRhelmetsS,TRvideoMetaS, on='game_play', how='left')#,TEvideoMetaS,on='game_play')\n\n### THEN JOIN WITH TRACKING DATA = TRAININGDATA\nTRtrackingS1 = TRtrackingS.copy()\ndel TRtrackingS1['game_play']\ndftr2 = dftr1.join(TRtrackingS1).dropna(axis=0)\n\ndftr21 = dftr2.copy()\n# convert to datetime and facilitate datasets/merge.\ndftr21['Tstart'] = pd.to_datetime(dftr21['start_time'])\ndftr21['Tend'] = pd.to_datetime(dftr21['end_time'])\ndftr21['Tsnap'] = pd.to_datetime(dftr21['snap_time'])\n\n\n# define time elapse(seconds) between start and end of video capture \n# convert to datetime and facilitate datasets/merge.\ndftr21['Tstart'] = pd.to_datetime(dftr21['start_time'])\ndftr21['Tend'] = pd.to_datetime(dftr21['end_time'])\n\n# define time elapse(seconds) between start and end of video capture \ndftr21['end_start_seconds'] = (dftr21['Tend'] - dftr21['Tstart']).dt.total_seconds()\n# define time elapse(seconds) between start and snapping of video capture \ndftr21['snap_start_seconds'] = (dftr21['Tsnap'] - dftr21['Tstart']).dt.total_seconds()\n# define time elapse(seconds) between end and snapping of video capture \ndftr21['end_snap_seconds'] = (dftr21['Tend'] - dftr21['Tsnap']).dt.total_seconds()\n\ndftr22 = dftr21.copy()\n\n### DATA TYPES CONVERSIONS OF SELECTED VARIABLES - Training Datasets\ndftr22['width']=dftr22['width'].astype('int32')\ndftr22['top']=dftr22['top'].astype('int32')\ndftr22['height']=dftr22['height'].astype('int32')\ndftr22['step']=dftr22['left'].astype('float32')\ndftr22['x_position']=dftr22['left'].astype('float32')\ndftr22['y_position']=dftr22['left'].astype('float32')\ndftr22['end_start_seconds']=dftr22['left'].astype('float32')\n\ndftr22S=dftr22[['game_play','width','view_y','team','position','x_position',\n               'y_position','end_start_seconds']]\ndftr22S.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:17.648068Z","iopub.execute_input":"2023-02-22T01:54:17.648519Z","iopub.status.idle":"2023-02-22T01:54:28.505652Z","shell.execute_reply.started":"2023-02-22T01:54:17.648487Z","shell.execute_reply":"2023-02-22T01:54:28.504477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TEST DATASETS**","metadata":{}},{"cell_type":"code","source":"### SELECT VARIABLES - Testing Datasets\nTEtrackingS = TEtracking[['game_play','step','team','position','x_position',\n                          'y_position']] # 'play_id',\nTEhelmetsS = TEhelmets[['game_play','view','left', 'width', 'top', 'height']]\nTEvideoMetaS = TEvideoMeta[['game_play','start_time', 'end_time','snap_time']]\n\n### MERGE HELMET AND VIDEO DATA\ndfte1 = pd.merge(TEhelmetsS,TEvideoMetaS, on='game_play', how='left')#,TEvideoMetaS,on='game_play')\n\n### THEN JOIN WITH TRACKING DATA = TRAININGDATA\nTEtrackingS1 = TEtrackingS.copy()\ndel TEtrackingS1['game_play']\ndfte2 = dfte1.join(TEtrackingS1).dropna(axis=0)\n\ndfte21 = dfte2.copy()\n# convert to datetime and facilitate datasets/merge.\ndfte21['Tstart'] = pd.to_datetime(dfte21['start_time'])\ndfte21['Tend'] = pd.to_datetime(dfte21['end_time'])\ndfte21['Tsnap'] = pd.to_datetime(dfte21['snap_time'])\n\n\n# define time elapse(seconds) between start and end of video capture \n# convert to datetime and facilitate datasets/merge.\ndfte21['Tstart'] = pd.to_datetime(dfte21['start_time'])\ndfte21['Tend'] = pd.to_datetime(dfte21['end_time'])\n\n# define time elapse(seconds) between start and end of video capture \ndfte21['end_start_seconds'] = (dfte21['Tend'] - dfte21['Tstart']).dt.total_seconds()\n# define time elapse(seconds) between start and snapping of video capture \ndfte21['snap_start_seconds'] = (dfte21['Tsnap'] - dfte21['Tstart']).dt.total_seconds()\n# define time elapse(seconds) between end and snapping of video capture \ndfte21['end_snap_seconds'] = (dfte21['Tend'] - dfte21['Tsnap']).dt.total_seconds()\n\ndfte22 = dfte21.copy()\n\n### DATA TYPES CONVERSIONS OF SELECTED VARIABLES - Testing Datasets\ndfte22['width']=dfte22['width'].astype('int32')\ndfte22['top']=dfte22['top'].astype('int32')\ndfte22['height']=dfte22['height'].astype('int32')\ndfte22['step']=dfte22['left'].astype('float32')\ndfte22['x_position']=dfte22['left'].astype('float32')\ndfte22['y_position']=dfte22['left'].astype('float32')\ndfte22['end_start_seconds']=dfte22['left'].astype('float32')\n\ndfte22S=dfte22[['game_play','width','view','team','position','x_position',\n               'y_position','end_start_seconds']]\ndfte22S.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:28.507510Z","iopub.execute_input":"2023-02-22T01:54:28.507819Z","iopub.status.idle":"2023-02-22T01:54:28.634098Z","shell.execute_reply.started":"2023-02-22T01:54:28.507790Z","shell.execute_reply":"2023-02-22T01:54:28.632996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:28.635710Z","iopub.execute_input":"2023-02-22T01:54:28.636035Z","iopub.status.idle":"2023-02-22T01:54:28.743522Z","shell.execute_reply.started":"2023-02-22T01:54:28.636003Z","shell.execute_reply":"2023-02-22T01:54:28.742267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TRAINLABELS DATASET**","metadata":{}},{"cell_type":"code","source":"### CONVERT DATA TYPES\ntrainlabels['step']=trainlabels['step'].astype('int32')\ntrainlabels['nfl_player_id_1']=trainlabels['nfl_player_id_1'].astype('int32')\ntrainlabels['contact']=trainlabels['contact'].astype('int32')\n\ntrainlabelsS = trainlabels.drop('datetime',axis=1)\ntrainlabelsS = trainlabels.drop('game_play',axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:28.745412Z","iopub.execute_input":"2023-02-22T01:54:28.745735Z","iopub.status.idle":"2023-02-22T01:54:29.357168Z","shell.execute_reply.started":"2023-02-22T01:54:28.745705Z","shell.execute_reply":"2023-02-22T01:54:29.356037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### RECONCILE TRAININGDATA WITH TRAINLABELS\n### DEFINE TRAINDATA for training the model\n\ndftr3 = dftr22S.join(trainlabelsS).dropna(axis=0)\nTRDATA = dftr3.drop('datetime', axis=1)\ndel TRDATA['contact_id']\n#del TRDATA['contact']\nTRDATA1 = TRDATA.loc[TRDATA['nfl_player_id_2'] != 'G']\nTRDATA2 = TRDATA.loc[TRDATA['nfl_player_id_2'] == 'G']\n\ndisplay(TRDATA[:3])\ndisplay(TRDATA.shape)\nprint('')\ndisplay(TRDATA1[:3])\ndisplay(TRDATA1.shape)\nprint('')\ndisplay(TRDATA2[:3])\ndisplay(TRDATA2.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:29.358762Z","iopub.execute_input":"2023-02-22T01:54:29.359522Z","iopub.status.idle":"2023-02-22T01:54:31.511477Z","shell.execute_reply.started":"2023-02-22T01:54:29.359489Z","shell.execute_reply":"2023-02-22T01:54:31.510480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### RECONCILE TESTINGDATA WITH TRAINLABELS\n### DEFINE TESTDATA for testing the model\n\ndfte3 = dfte22S.join(trainlabelsS).dropna(axis=0)\nTEDATA = dfte3.drop('datetime', axis=1)\ndel TEDATA['contact_id']\ndel TEDATA['contact']\nTEDATA1 = TEDATA.loc[TEDATA['nfl_player_id_2'] != 'G']\nTEDATA2 = TEDATA.loc[TEDATA['nfl_player_id_2'] == 'G']\n\ndisplay(TEDATA[:3])\ndisplay(TEDATA.shape)\nprint('')\ndisplay(TEDATA1[:3])\ndisplay(TEDATA1.shape)\nprint('')\ndisplay(TEDATA2[:3])\ndisplay(TEDATA2.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:31.513249Z","iopub.execute_input":"2023-02-22T01:54:31.513677Z","iopub.status.idle":"2023-02-22T01:54:31.617988Z","shell.execute_reply.started":"2023-02-22T01:54:31.513635Z","shell.execute_reply":"2023-02-22T01:54:31.616990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ML MODEL**","metadata":{"execution":{"iopub.status.busy":"2023-02-21T20:26:50.948553Z","iopub.execute_input":"2023-02-21T20:26:50.949038Z","iopub.status.idle":"2023-02-21T20:26:50.957852Z","shell.execute_reply.started":"2023-02-21T20:26:50.949002Z","shell.execute_reply":"2023-02-21T20:26:50.955904Z"}}},{"cell_type":"markdown","source":"**LOGISTIC REGRESSION MODEL**","metadata":{}},{"cell_type":"code","source":"# CHECK THE DISTRIBUTION OF THE 'CONTACT' VARIABLE\nTRDATA['contact'].value_counts()/TRDATA.shape[0]\n\n## This distribution suggests the need to use of weighted logistic model","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:31.619806Z","iopub.execute_input":"2023-02-22T01:54:31.620108Z","iopub.status.idle":"2023-02-22T01:54:31.638988Z","shell.execute_reply.started":"2023-02-22T01:54:31.620080Z","shell.execute_reply":"2023-02-22T01:54:31.637796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import model and matrics\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score, RepeatedStratifiedKFold, StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, confusion_matrix,roc_curve, roc_auc_score, precision_score, recall_score, precision_recall_curve\nfrom sklearn.metrics import f1_score\n\nTRDATA['nfl_player_id_2'] = TRDATA['nfl_player_id_2'].replace(['G'], 1.0)\n\n# split dataset into x,y\nx = TRDATA[['game_play','width','x_position','y_position','end_start_seconds','step','nfl_player_id_1','nfl_player_id_2']]# .drop('contact',axis=1)\ny = TRDATA['contact']\n###\nscaler = StandardScaler()#.set_output(transform=\"pandas\")\nscaled_x  = scaler.fit(x)\n#x =scaled_x \n# define class weights\nw = {0:2, 1:98}\n# train-test split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.3, random_state=123)\nscaled_x_train = scaler.fit_transform(x_train)\n# define model\nLogReg = LogisticRegression(random_state=13, class_weight=w, max_iter=10000)\n# fit it\nLogReg.fit(x_train, y_train)\n# test\ny_pred = LogReg.predict(x_test)\n# performance\nprint(f'Accuracy Score: {accuracy_score(y_test,y_pred)}')\nprint(f'Confusion Matrix: \\n{confusion_matrix(y_test, y_pred)}')\nprint(f'Area Under Curve: {roc_auc_score(y_test, y_pred)}')\nprint(f'Recall score: {recall_score(y_test,y_pred)}')","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:31.640278Z","iopub.execute_input":"2023-02-22T01:54:31.640615Z","iopub.status.idle":"2023-02-22T01:54:38.818041Z","shell.execute_reply.started":"2023-02-22T01:54:31.640585Z","shell.execute_reply":"2023-02-22T01:54:38.816800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del TEDATA['view']\ndel TEDATA['team']\ndel TEDATA['position']\nTEDATA['nfl_player_id_2'] = TEDATA['nfl_player_id_2'].replace(['G'], 1.0)\ny_pred_TEDATA = LogReg.predict(TEDATA)\ndf_y_pred_TEDATA=pd.DataFrame(y_pred_TEDATA)\ndf_y_pred_TEDATA.rename(columns={0:'contact'}, inplace=True)\ndf_y_pred_TEDATA","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:38.819800Z","iopub.execute_input":"2023-02-22T01:54:38.820207Z","iopub.status.idle":"2023-02-22T01:54:38.863142Z","shell.execute_reply.started":"2023-02-22T01:54:38.820165Z","shell.execute_reply":"2023-02-22T01:54:38.861696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RECREATE 'contact_id'\nTEDATA['game_play']=TEDATA['game_play'].astype('str')\nTEDATA['step']=TEDATA['step'].astype('str')\nTEDATA['nfl_player_id_1']=TEDATA['nfl_player_id_1'].astype('str')\nTEDATA['nfl_player_id_2']=TEDATA['nfl_player_id_2'].astype('str')\n\nTEDATA['contact_id'] = TEDATA['game_play'] + \"_\" + TEDATA['step'] + '_' + TEDATA['nfl_player_id_1'] + '_' + TEDATA['nfl_player_id_2']\ndisplay(TEDATA[:3])","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:38.869848Z","iopub.execute_input":"2023-02-22T01:54:38.870481Z","iopub.status.idle":"2023-02-22T01:54:38.978602Z","shell.execute_reply.started":"2023-02-22T01:54:38.870421Z","shell.execute_reply":"2023-02-22T01:54:38.977435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FOR SUBMISSION\nSubmission = pd.DataFrame()\nSubmission['contact_id'] = TEDATA[['contact_id']] \nSubmission['contact'] = df_y_pred_TEDATA['contact']\n#Submission = Submission.dropna()#, inplace=True\ndisplay(Submission[:3])","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:38.979967Z","iopub.execute_input":"2023-02-22T01:54:38.980313Z","iopub.status.idle":"2023-02-22T01:54:38.998400Z","shell.execute_reply.started":"2023-02-22T01:54:38.980281Z","shell.execute_reply":"2023-02-22T01:54:38.997252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VIEW THE FIRST # ROWS OF THE SUBMISSION \ndisplay(Submission[:3])\ndisplay(Submission.columns)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:38.999761Z","iopub.execute_input":"2023-02-22T01:54:39.000095Z","iopub.status.idle":"2023-02-22T01:54:39.014281Z","shell.execute_reply.started":"2023-02-22T01:54:39.000064Z","shell.execute_reply":"2023-02-22T01:54:39.013105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SAVE SUBMISSION TO FILE\nSubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-22T01:54:39.015484Z","iopub.execute_input":"2023-02-22T01:54:39.015880Z","iopub.status.idle":"2023-02-22T01:54:39.043995Z","shell.execute_reply.started":"2023-02-22T01:54:39.015851Z","shell.execute_reply":"2023-02-22T01:54:39.043183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}