{"cells":[{"metadata":{"_uuid":"6ea093ecb2f0abbd67b724ec763c0f3e41d3568f"},"cell_type":"markdown","source":"# Methodology\nSee slides for more details:\nhttps://drive.google.com/file/d/1A-TjTK5b0AHf47H9Pk1rZHMPsc51f4lZ/view?usp=sharing\n\n* Logistic Regression model\n * 6459 punts (36 of which had a concussion)\n* Categorical features\n * Punt from yard line\n * Punt to yard line\n * Return yards\n * Punt yards\n * Return team personel\n* Predict plays with a concussion and learn insight from coefficient\n* Explore impacts on game dynamics of potential rule changes (more/less likely to have a long return, no return attempt, etc.)\n* Use next gen stats (max speed/distance traveled) to better understand insights\n* Propose **actionable** rule changes that maintain integrity\n * Leaned heavily on prior changes done to kickoff as those are proven to be actionable:\n  * https://profootballtalk.nbcsports.com/2018/05/16/nfl-to-vote-on-new-kickoff-rules-limiting-full-speed-collisions/\n  * http://www.espn.com/blog/nflnation/post/_/id/287320/the-nfl-rule-tweaks-saving-kickoff-from-extinction\n* Considered unintended consequences/risks on strategy/safety of proposed ideas (see slides)\n\n# Proposal\n1. Max of one punt return player allowed beyond 15 yards at time of snap\n1. Max of two return team players outside of the numbers \n1. Max of 8 return team players on the line of scrimmage"},{"metadata":{"_uuid":"e0e9a97e1f0818c8b91fe2cdc66804044237d532"},"cell_type":"markdown","source":"## Load data"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom datetime import datetime\nimport datetime\nfrom sklearn.preprocessing import StandardScaler\nfrom datetime import timedelta\nimport re\nfrom scipy import sparse\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import GridSearchCV","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f574587fc55e73c93eb66102d6b08bafc405fd0"},"cell_type":"code","source":"injuries = pd.read_csv('../input/video_review.csv', delimiter=',')\nplays = pd.read_csv('../input/play_information.csv', delimiter=',')\npositions = pd.read_csv('../input/player_punt_data.csv', delimiter=',')\nroles = pd.read_csv('../input/play_player_role_data.csv', delimiter=',')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fa46dd30e3e20c8683443549bdf8ba7722ec32ea"},"cell_type":"markdown","source":"# Enhance data"},{"metadata":{"_uuid":"4faf629a19fd86655e1e6e7924ca241938c29d82"},"cell_type":"markdown","source":"### Categorize roles"},{"metadata":{"trusted":true,"_uuid":"5224322c8d6e07ad3f85549230234a44f6824f34"},"cell_type":"code","source":"# group positions into groups\n\npunt_return = {\n'role_return_line':['PDL1','PDL2','PDL3','PDL4','PDL5','PDL6','PDM','PDR1','PDR2','PDR3','PDR4','PDR5','PDR6'],\n'role_return_lead_blocker':['PFB'],\n'role_return_linebacker':['PLL','PLL1','PLL2','PLL3','PLR','PLR1','PLR2','PLR3','PLM','PLM1'],\n'role_return_returner':['PR'],\n'role_return_jammer':['VR','VR','VL','VLi','VLo','VRi','VRo']}\n\npunt_team = {\n'role_punt_line':['PLG','PLS','PLT','PRG','PRT'],\n'role_punt_wings':['PLW','PRW'],\n'role_punt_punter':['P'],\n'role_punt_protector':['PPR','PC','PPL','PPLi','PPLo','PPRi','PPRo'],\n'role_punt_gunner':['GL','GLi','GLo','GR','GRi','GRo']}\n\nroles['punt_or_return'] = np.nan\nroles['role_group'] = np.nan\nfor k,v in punt_return.items():\n    roles[k] = np.where(roles['Role'].isin(v), 1, 0)\n    roles['punt_or_return'] = np.where(roles['Role'].isin(v), 'return', roles['punt_or_return'])\n    roles['role_group'] = np.where(roles['Role'].isin(v), k, roles['role_group'])\nfor k,v in punt_team.items():\n    roles[k] = np.where(roles['Role'].isin(v), 1, 0)\n    roles['punt_or_return'] = np.where(roles['Role'].isin(v), 'punt', roles['punt_or_return'])\n    roles['role_group'] = np.where(roles['Role'].isin(v), k, roles['role_group'])\nroles.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65de8c65d6ff80cb201cb56871919e764789afa6"},"cell_type":"code","source":"# create a categorical for each role group based on whether there were more/less players relative a typical punt\n\nroles_per_play = roles[['GameKey','PlayID','punt_or_return','role_group',\n    'role_return_line','role_return_lead_blocker','role_return_linebacker','role_return_returner','role_return_jammer',\n    'role_punt_line','role_punt_punter','role_punt_protector','role_punt_gunner','role_punt_wings']].groupby(['GameKey','PlayID']).sum()\n\nfor group in list(punt_return)+list(punt_team):\n    median = int(roles_per_play[group].median())\n    # hack for now since median is 2 but most common 2,3,4 - maybe should have used pd.cut\n    if group == 'role_return_jammer':\n        median = 3\n    print(group, median, roles_per_play[group].value_counts().to_dict())\n    roles_per_play[group] = pd.cut(roles_per_play[group], [-1, median-0.5, median + 0.5, 100], labels=[str(median-1)+'_or_less',str(median),str(median+1)+'_or_more'])\n    print(roles_per_play[group].value_counts().to_dict())\nroles_per_play.head()\n# The punt team doesn't seem to have much variation, however the return team has some flexability which might be useful to take advantage of!","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d6a388820f1c4ea228eb0e47aeda7c1e1c72ce32"},"cell_type":"markdown","source":"### Merge data"},{"metadata":{"trusted":true,"_uuid":"3c4fef2c2f7bad6bfd74bc1dd25dd380750eb83e","scrolled":true},"cell_type":"code","source":"# one row per play\ndf = plays.copy()\n\n# merge with concussions\ndf = df.merge(injuries[['GameKey','PlayID','Player_Activity_Derived','Turnover_Related','Primary_Impact_Type',\n                             'GSISID','Friendly_Fire']],how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\n\n# remove plays which dont actually end up with a punt \ndf = df[~df['PlayDescription'].str.contains('BLOCKED|Aborted|False Start| pass |Delay of Game')]\n# note that this removes a concussion play sample: \"J.Ryan up the middle to LA 47 for 26 yards. FUMBLES, recovered by SEA-N.Thorpe at LA 40. SEA-J.Ryan was injured during the play.  Los Angeles challenged the loose ball recovery ruling, and the play was Upheld. The ruling on the field stands. (Timeout #2.)']\"\ndf = df[df['PlayDescription'].str.contains('punts')]\nprint(len(df), len(plays))\n\n# merge to get role (for the injured player)\ndf = df.merge(roles[['GameKey','PlayID','GSISID','Role','punt_or_return','role_group']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n\n# merge to get number of each role on the play\ndf = df.merge(roles_per_play,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\n\n# merge with positions (for injured player) - note that positions has dups\ndf = df.merge(positions[['GSISID','Position']].drop_duplicates(subset=['GSISID']),how='left',left_on=['GSISID'],right_on=['GSISID'])\n\n# hack - removing a bunch of text that made my life more difficult to parse\ndf['cleaned_description'] = df['PlayDescription'].str.replace('FUMBLES.*|recovers.*|RECOVERED.*|recovered.*|Officially.*|.*REVERSED|challenged the kick downed|downed the ball at the .*, but was ruled out-of-bounds|for the remainder|for a concussion','')\n\n# whether or not there was a concussion\ndf['had_concussion'] = np.where(df['Primary_Impact_Type'].isnull(), 0, 1)\n\n# tagging the event types\nevent_types = ['downed','Touchback','fair catch','for -*[0-9]* yard','for no gain','MUFFS','out of bounds\\.']\nfor text in event_types:\n    df[text] = np.where(df['cleaned_description'].str.contains(text), 1, 0)\n    print(text, df[text].sum(), df[df[text]>0]['had_concussion'].sum())\n    \n# make sure all rows only have one tagged event type\nprint('dropping invalid tagged events=', df[df[event_types].sum(axis=1)!=1]['PlayDescription'].values)\ndf = df[df[event_types].sum(axis=1)==1]\n\nprint(len(df), len(plays), df.columns)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c44d7f0f9e43acd046c88ad75a77a8d5e3abc890"},"cell_type":"markdown","source":"### Add features"},{"metadata":{"trusted":true,"_uuid":"eeea71f1be2e9774b9b45587c794190465a3270f","scrolled":true},"cell_type":"code","source":"# punts yards\ndf['punt_yards'] = df['PlayDescription'].str.replace('.* punts ','').str.replace(' yard.*','').astype(int)\n\n# return yards\ndf['return_yards'] = np.where(df['for -*[0-9]* yard']==1, \n                                  df['cleaned_description'].str.replace('.* for| yard.*|-yds.*',''), \n                                  np.where(df['for no gain']==1, 0, np.nan)).astype(float)\n\n# which side of field punting from\ndf['punt_from_own'] = df.apply(lambda row: row['Poss_Team'] in row['YardLine'], axis=1)\n\n# yard being punted from (0->100 where 0 is the punting teams own end zone)\ndf['punt_from'] = df['YardLine'].str.replace('.* ','').astype(int)\ndf['punt_from'] = np.where(df['punt_from_own'], df['punt_from'], 100-df['punt_from'])\ndf['line_of_scrimmage'] = df['punt_from']\n\n# yard being punted to (0->100 where 0 is the return teams own end zone)\ndf['punt_to'] = 100 - (df['punt_from'] + df['punt_yards'].astype(float))\n\n# whether or not a return was attempted\ndf['no_return_attempted'] = np.where((df['downed']==1)|(df['Touchback']==1)|(df['fair catch']==1)|(df['out of bounds\\.']==1), 1, 0)\n\n# convert to categoricals\ndf['punt_yards'] = pd.cut(df['punt_yards'], [-1,40,46,52,100], labels=['40_or_less','40_to_46','46_to_52','52_or_more'])\ndf['punt_from'] = pd.cut(df['punt_from'], [-1,10,20,40,100], labels=['within_10','10_to_25','25_to_40','40_or_more'])\ndf['punt_to'] = pd.cut(df['punt_to'], [-1,10,20,40,100], labels=['within_10','10_to_25','25_to_40','40_or_more'])\ndf['return_yards'] = pd.cut(df['return_yards'],[-100,-0.5,5.5,15,100], labels=['negative','0_to_5','5_to_15','more_than_15'])\ndf['return_yards'] = np.where(df['no_return_attempted']>0,'no_return_attempted',df['return_yards'])\ndf['return_yards'] = np.where(df['MUFFS']>0,'muffed_punt',df['return_yards'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"276f311f75e119c3ad1ebe6486ab81f96e5184f6"},"cell_type":"markdown","source":"## High level look"},{"metadata":{"trusted":true,"_uuid":"f30986bba99d23befe22a501a42bd3789f7c5d9e"},"cell_type":"code","source":"print(df.groupby(['had_concussion'])['had_concussion'].count().to_frame())\nprint(df.groupby(['punt_or_return','Player_Activity_Derived'])['Player_Activity_Derived'].count().to_frame())\nprint(df.groupby(['punt_or_return','Player_Activity_Derived'])['Player_Activity_Derived'].count().to_frame())\nprint(df.groupby(['punt_or_return','role_group'])['role_group'].count().to_frame())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"811716a1260630b6231500df4963db43354a4b38"},"cell_type":"markdown","source":"Observations:\n* 26 of 36 of concussed players are from the punting team\n* Of the 10 return team concussions, 5 are from being tackled\n* Larger players from the punt team (lineman/wings/protector) are 21 of the 36 concussions\n* Surprisingly only 5 of the concussions are on the returner\n\nNeed to find a way to limit the speed of the bigger players from the punt team! This will reduce high speed collisions caused by the punting team."},{"metadata":{"_uuid":"5b83602be37537962c3f6e6e8c0d23baa4613d9b"},"cell_type":"markdown","source":"## Model"},{"metadata":{"trusted":true,"_uuid":"cf9b819b00452b6c9f08b8e59c1e9432a2018588","scrolled":true},"cell_type":"code","source":"X = df[['punt_yards','punt_from','punt_to','return_yards','role_return_lead_blocker', 'role_return_linebacker','role_return_line', 'role_return_jammer']]\n\nfor col in X.columns:\n    print(col, X[col].value_counts().to_dict())\n\nX = pd.get_dummies(X)\nX = X.loc[:, (X != 0).any(axis=0)]\ny = df['had_concussion']\nprint(X.columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30b74a77c712f3ec288f64fcd8b766eeb52410e1","_kg_hide-output":true},"cell_type":"code","source":"model = LogisticRegression()\nmodel = model.fit(X, y)\n\ncoeff = pd.DataFrame(model.coef_.T, X.columns, columns=['coeff']).sort_values('coeff',ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"931a1390f7e325bd0bd2bec992b0ced137fd3aed"},"cell_type":"markdown","source":"## Results"},{"metadata":{"trusted":true,"_uuid":"9d388b40445940801ac7a2f14d58ddc8a0d44d9a"},"cell_type":"code","source":"print(coeff)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e23b66ed9aec2f30d306223846101cad927e7d3"},"cell_type":"code","source":"for prefix in ['role_return_lead_blocker','role_return_line_','role_return_lineb','role_return_jammer','punt_to','punt_from','return_yards','punt_yards']:\n    print(coeff[coeff.index.str.startswith(prefix)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"044edc0d6aac12fb982bd8d3596bf57de420cd92"},"cell_type":"markdown","source":"### Key takeaways\n* Longer returns = more concussions (more players running around)\n* No return (touchback, fair catch) eliminates concussions\n* Punts closer to the middle of the field are more dangerous (correlates with where punting from)\n * Most dangerous punts are 40->46 yards (not quite out kicking the coverage but not fair caught)\n* No lead blocker = less concusions\n* 2 or more linebackers = less concussions\n* 5 or less lineman = more concussions\n* 7 or more lineman = more concussions\n* 3 or more jammers = more concussions\n* Alignment / personel matter\n\n***More analysis in slides.***\n\n**Unintended consequences notes/open questions:**\n* Reducing jammers could increase gunners getting down the field unblocked for big hits on returner\n* Will this result in more/less faster/slower players - what does it mean? how does personel change?\n* More or less returns?\n* What safety risks might increase?\n* What strategy will change?\n* How far downfield are guys getting? Distance traveled?"},{"metadata":{"_uuid":"c905d59602e221e6aa157459787d364c8d00ee99"},"cell_type":"markdown","source":"## Impacts\nWe can run similar models to predict other attributes (large return, no return attempted) to see how changing personel might change game dynamics. Other things to explore might be penalties, turnovers, etc. Keep in mind that by including other attributes such as field position and punt length we are able to control for situation."},{"metadata":{"trusted":true,"_uuid":"f79b2c1310dc56dc73c53e71ff2ad01cfc005e9d"},"cell_type":"code","source":"# predict large return\nX = df[['punt_yards','punt_from','punt_to','role_return_lead_blocker', 'role_return_linebacker','role_return_line', 'role_return_jammer']]\nX = pd.get_dummies(X)\nX = X.loc[:, (X != 0).any(axis=0)]\ny = np.where(df['return_yards']=='more_than_15',1,0)\n\nmodel = LogisticRegression()\nmodel = model.fit(X, y)\n\ncoeff = pd.DataFrame(model.coef_.T, X.columns, columns=['coeff']).sort_values('coeff',ascending=False)\nprint(coeff[coeff.index.str.contains('role')])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"53dd360de5d3fb1458992c7029e854f13234649a"},"cell_type":"markdown","source":"Potential impacts on large returns:\n* Less jammers = less large returns\n* More line backers = less large returns\n* No lead blocker = more large returns\n* <7 lineman = more large returns (slightly)\n* Overall = wash?"},{"metadata":{"trusted":true,"_uuid":"198c82559aa52415a4f32e415e2ef1c55676910c"},"cell_type":"code","source":"# predict no return attempted\nX = df[['punt_yards','punt_from','punt_to','role_return_lead_blocker', 'role_return_linebacker','role_return_line', 'role_return_jammer']]\nX = pd.get_dummies(X)\nX = X.loc[:, (X != 0).any(axis=0)]\ny = np.where(df['return_yards']=='no_return_attempted',1,0)\n\nmodel = LogisticRegression()\nmodel = model.fit(X, y)\n\ncoeff = pd.DataFrame(model.coef_.T, X.columns, columns=['coeff']).sort_values('coeff',ascending=False)\nprint(coeff[coeff.index.str.contains('role')])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d8b02a264b494f9eb708d9f2766469fde533780f"},"cell_type":"markdown","source":"Potential impacts on no return attempted (fair catch, touchback):\n* Less jammers = a lot less returns attempted\n* More line backers = more returns attemped\n* No lead blocker = a lot less returns attempted\n* <7 lineman = less returns attempted\n* Overall = less returns attempted"},{"metadata":{"trusted":true,"_uuid":"156cc9faf76deb2765de6d75d61a050cd032f4dc"},"cell_type":"code","source":"# predict muffed punt\nX = df[['punt_yards','punt_from','punt_to','role_return_lead_blocker', 'role_return_linebacker','role_return_line', 'role_return_jammer']]\nX = pd.get_dummies(X)\nX = X.loc[:, (X != 0).any(axis=0)]\ny = np.where(df['return_yards']=='muffed_punt',1,0)\n\nmodel = LogisticRegression()\nmodel = model.fit(X, y)\n\ncoeff = pd.DataFrame(model.coef_.T, X.columns, columns=['coeff']).sort_values('coeff',ascending=False)\nprint(coeff[coeff.index.str.contains('role')])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"505837eb9471d2c9b435a6c8e556eea9f22616bd"},"cell_type":"markdown","source":"Potential impacts on muffed punts:\n* Less jammers = less?\n* More line backers = more?\n* No lead blocker = more\n* <7 lineman = less\n* Overall = wash?"},{"metadata":{"trusted":true,"_uuid":"8fd4d47800182fff1d84a5bce364b9bcd8ec8275"},"cell_type":"markdown","source":"## Next Gen Stats data\nBelow is where I will play around withi next-gen data to see if we can better understand observations and show why the features are having the effect they do\n\nThis code was more exploratory and I dont fully trust some of the data without further sanity checking of quality/caveats and more properly cleaning"},{"metadata":{"trusted":true,"_uuid":"005ca17c48bcbb0c72e12a9bfb10a9ede3677f39"},"cell_type":"code","source":"# load next gen stats to calculate some different metrics\nnext_gen = pd.DataFrame()\nfor f in ['NGS-2016-pre','NGS-2016-post','NGS-2016-reg-wk1-6','NGS-2016-reg-wk7-12','NGS-2016-reg-wk13-17',\n    'NGS-2017-pre','NGS-2017-post','NGS-2017-reg-wk1-6','NGS-2017-reg-wk7-12','NGS-2017-reg-wk13-17']:\n    cur = pd.read_csv('../input/%s.csv'%(f), delimiter=',')\n    \n    # time when the alignment should be set\n    starting_event = cur[cur.Event.isin(['line_set','ball_snap'])].sort_values('Time').drop_duplicates(subset=['GameKey','PlayID'])[['GameKey','PlayID','Time']].rename(columns={'Time':'snap_start_time'})\n    \n    # throw out pre-snap data\n    cur = cur.merge(starting_event,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\n    cur = cur[cur['Time'] >= cur['snap_start_time']]\n    \n    cur = cur.merge(df[['GameKey','PlayID','line_of_scrimmage']],how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID']).dropna(subset=['line_of_scrimmage'])\n    \n    # use the earliest datapoint for each play/player to get starting player location\n    initial_location = cur.sort_values('Time').drop_duplicates(subset=['GameKey','PlayID','GSISID'])[['GameKey','PlayID','GSISID','x','y']].rename(columns={'x':'x_initial','y':'y_initial'})\n    cur = cur.merge(initial_location,how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n    \n    # add punt timestamp\n    punt_time = cur[cur['Event']=='punt'].sort_values('Time').drop_duplicates(subset=['GameKey','PlayID'])[['GameKey','PlayID','Time']].rename(columns={'Time':'punt_timestamp'})\n    \n    # now calculate when player crosses line of scrimmage\n    cross_los = cur[['GameKey','PlayID','GSISID','line_of_scrimmage','x_initial','Time']].drop_duplicates()\n    # if player started on left side of line of scrimmage then see when x > los+1, otherwise see when x < los-1\n    cross_los = cross_los[((cur['x_initial']-10 < cur['line_of_scrimmage'])&(cur['x']-10 > cur['line_of_scrimmage']+1))|\n                          ((cur['x_initial']-10 > cur['line_of_scrimmage'])&(cur['x']-10 < cur['line_of_scrimmage']-1))]\n    cross_los = cross_los.sort_values('Time').drop_duplicates(subset=['GameKey','PlayID','GSISID'])[['GameKey','PlayID','GSISID','Time','line_of_scrimmage']].rename(columns={'Time':'time_crossed_los'})\n    \n    # lets ignore outliers like this:\n    #  75.639999  51.799999  2016-08-12 01:45:02.800  0.00\n    #  75.639999  51.799999  2016-08-12 01:45:02.900  0.00\n    #  71.739998  53.310001  2016-08-12 01:45:03.000  4.18\n    #  75.750000  51.740002  2016-08-12 01:44:58.500  0.00\n    #  75.750000  51.740002  2016-08-12 01:44:58.600  0.00\n    cur = cur[cur['dis'] < 1.4667] # this translates to roughly 30 mph\n    \n    max_distance = cur.groupby(['GameKey','PlayID','GSISID'])['dis'].max().to_frame().reset_index().rename(columns={'dis':'max_distance'})\n    \n    # calculate max_mph: 1 yard/second = 2.04545 miles/hour, then multiply by 10 since data points are tenth of a second\n    max_distance['max_mph'] = max_distance['max_distance'] * 2.04545 * 10\n    cur_next_gen = max_distance[['GameKey','PlayID','GSISID','max_mph']]\n    \n    # calculate total distance traveled on play\n    total_distance = cur.groupby(['GameKey','PlayID','GSISID'])['dis'].sum().to_frame().reset_index().rename(columns={'dis':'total_distance'})\n    cur_next_gen = cur_next_gen.merge(total_distance,how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n\n    # merge other metrics\n    cur_next_gen = cur_next_gen.merge(initial_location,how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n    cur_next_gen = cur_next_gen.merge(cross_los,how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n    cur_next_gen = cur_next_gen.merge(punt_time,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\n    \n    next_gen = pd.concat([next_gen, cur_next_gen])\n\n# http://static.nfl.com/static/content/public/image/rulebook/pdfs/12_Rule9_Scrimmage_Kick.pdf\n# During a kick from scrimmage, only the end men (eligible receivers) on the line of scrimmage at\n# the time of the snap, or an eligible receiver who is aligned or in motion behind the line and is more than\n# one yard outside the end man, are permitted to advance more than one yard beyond the line before the\n# ball is kicked. \nnext_gen['crossed_line_before_punt'] = np.where(next_gen['punt_timestamp']>next_gen['time_crossed_los'],1,0)\nnext_gen['outside_numbers'] = np.where((next_gen['y_initial']<12)|(next_gen['y_initial']>(53.3-12)), 1, 0)\nnext_gen['within_5_of_los'] = np.where(np.abs((next_gen['x_initial']-10)-next_gen['line_of_scrimmage'])<=5, 1, 0)\nnext_gen['within_10_of_los'] = np.where(np.abs((next_gen['x_initial']-10)-next_gen['line_of_scrimmage'])<=10, 1, 0)\nnext_gen['within_15_of_los'] = np.where(np.abs((next_gen['x_initial']-10)-next_gen['line_of_scrimmage'])<=15, 1, 0)\n\nnext_gen['max_mph'].hist()\nplt.show()\nnext_gen['total_distance'].hist()\nplt.show()\nnext_gen['x_initial'].hist()\nplt.show()\nnext_gen['y_initial'].hist()\nplt.show()\nnext_gen.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"924700f360274fc33658a6dd54377c55e290ac40"},"cell_type":"markdown","source":"Lets do some sanity checking..."},{"metadata":{"trusted":true,"_uuid":"8507aa97dec20b964d7e792fb6c71af13c345947"},"cell_type":"code","source":"print(next_gen.groupby(['GameKey','PlayID'])['x_initial'].count().value_counts().head(10))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"453ebf3e1b6780c3f2fbd6bf9c128ece64d4302b"},"cell_type":"markdown","source":"What!?!? There is most commonly data from 40+ players for each play! Must be players getting off the field from previous play?\n\nLets try filtering all of the players who started the play off the field..."},{"metadata":{"trusted":true,"_uuid":"76c6f583a00ea3ec9cecc36c9de5cac427870df5"},"cell_type":"code","source":"print(next_gen[~((next_gen['x_initial']<0)|(next_gen['x_initial']>53.3))].groupby(['GameKey','PlayID'])['x_initial'].count().value_counts().head(10))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af715e5034b300dff3b0f48290c9e2d33d7bcc9c"},"cell_type":"markdown","source":"Better but still not perfect!\n\nConclusion - this next gen data is going to be messy - if I dont concistently have data from exactly 22 players on each play then it will be difficult to trust some of the conclusions. This will need more time to deal with...\n\nAt the very least I can pull out some speed / distance metrics on individual players in a play"},{"metadata":{"trusted":true,"_uuid":"92c7aa503da87f2721a56330e6a4ef3094a330b6","_kg_hide-output":true,"_kg_hide-input":true},"cell_type":"code","source":"# Ideally I'd feed this data into the model but I dont trust it enough.\nfiltered = next_gen[~((next_gen['x_initial']<0)|(next_gen['x_initial']>53.3))]\nfiltered = filtered.merge(roles[['GameKey','PlayID','GSISID','punt_or_return']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n\n# limit to punt return team\nfiltered = filtered[filtered['punt_or_return']=='return']\n# limit to plays with exactly 11\nhave_11 = filtered.groupby(['GameKey','PlayID'])['punt_or_return'].count().to_frame().reset_index()\n#have_11 = have_11[have_11['punt_or_return']==11][['GameKey','PlayID']]\nfiltered = have_11.merge(filtered,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\n\nfiltered = filtered.merge(df[['GameKey','PlayID','had_concussion']],how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\nnext_gen_play_stats = filtered[['outside_numbers','within_5_of_los','within_10_of_los','within_15_of_los','GameKey','PlayID','had_concussion']].groupby(['GameKey','PlayID','had_concussion']).sum()#agg({'count','sum'})\nnext_gen_play_stats.groupby('had_concussion').agg({'mean','count'})#['outside_numbers'].count()#.mean()#gg({'mean','count'})#merge(roles[['GameKey','PlayID','GSISID','role_group','punt_or_return']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9370574d1c95b98127cf5e23691f40e7943d32f6","_kg_hide-output":true,"_kg_hide-input":true},"cell_type":"code","source":"#look at crossed before punt\n\nfiltered = next_gen[~((next_gen['x_initial']<0)|(next_gen['x_initial']>53.3))]\nfiltered = filtered.merge(roles[['GameKey','PlayID','GSISID','punt_or_return']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n\n# limit to punt return team\nfiltered = filtered[filtered['punt_or_return']=='punt']\n# limit to plays with exactly 11\nhave_11 = filtered.groupby(['GameKey','PlayID'])['punt_or_return'].count().to_frame().reset_index()\nhave_11 = have_11[have_11['punt_or_return']==11][['GameKey','PlayID']]\nfiltered = have_11.merge(filtered,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\n\nfiltered = filtered.merge(df[['GameKey','PlayID','had_concussion']],how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\nnext_gen_play_stats = filtered[['crossed_line_before_punt','outside_numbers','within_5_of_los','within_10_of_los','within_15_of_los','GameKey','PlayID','had_concussion']].groupby(['GameKey','PlayID','had_concussion']).sum()#agg({'count','sum'})\nnext_gen_play_stats.groupby('had_concussion').agg({'mean','count'})#['outside_numbers'].count()#.mean()#gg({'mean','count'})#merge(roles[['GameKey','PlayID','GSISID','role_group','punt_or_return']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"262cc49eeaf16cc2ef381fc23948b0b13b5a326d"},"cell_type":"markdown","source":"### Max speed of players involved in concussion"},{"metadata":{"trusted":true,"_uuid":"4267b8e1526e1fb7a6bf7814a8a51725ea037c71"},"cell_type":"code","source":"concussion_players_speed = injuries.merge(next_gen[['GameKey','PlayID','GSISID','max_mph']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\nconcussion_players_speed['Primary_Partner_GSISID'] = concussion_players_speed['Primary_Partner_GSISID'].replace('Unclear',-1).fillna(-1).astype(int)\nconcussion_players_speed['max_mph_partner'] = concussion_players_speed[['GameKey','PlayID','Primary_Partner_GSISID']].merge(next_gen[['GameKey','PlayID','GSISID','max_mph']],how='left',\n    left_on=['GameKey','PlayID','Primary_Partner_GSISID'],right_on=['GameKey','PlayID','GSISID'])['max_mph'].values\n\nplt.show()\nax = plt.axes()\nsns.scatterplot(x=\"max_mph_partner\", y=\"max_mph\", data=concussion_players_speed, ax=ax)\nax.set_title('Max speed (MPH) of players involved in a concussion')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"585ab3cd46bb5291b52df5e4ee21629997ed09d3"},"cell_type":"markdown","source":"### Join next gen with roles data"},{"metadata":{"trusted":true,"_uuid":"8e52bcb70051d3530ea29445a9d8896bbe22cd84"},"cell_type":"code","source":"next_gen_with_roles = next_gen.merge(roles[['GameKey','PlayID','GSISID','role_group','punt_or_return']],how='left',left_on=['GameKey','PlayID','GSISID'],right_on=['GameKey','PlayID','GSISID'])\n\n# for each play/role_group - get the max_mph/avg_distance across the players within that role_group\nmax_mph_per_play_role = next_gen_with_roles.groupby(['GameKey','PlayID','role_group','punt_or_return'])['max_mph'].max().to_frame().reset_index()\navg_distance_per_play_role = next_gen_with_roles.groupby(['GameKey','PlayID','role_group','punt_or_return'])['total_distance'].mean().to_frame().reset_index()\n\n# now merge with roles_per_play data\nmax_mph_per_play_role = roles_per_play.merge(max_mph_per_play_role,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])\navg_distance_per_play_role = roles_per_play.merge(avg_distance_per_play_role,how='left',left_on=['GameKey','PlayID'],right_on=['GameKey','PlayID'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08d91d581b96c55e3d939cf246ab64301fa2746e"},"cell_type":"markdown","source":"### Impact on punt team Max speed by changing personel"},{"metadata":{"trusted":true,"_uuid":"070675c308f2dece395fcf3f0cddfa71cd853333"},"cell_type":"code","source":"# show impact on punt team max_mph by changing a return team position group\nfor role_group in ['role_return_line','role_return_linebacker','role_return_lead_blocker','role_return_jammer']:\n    filtered = max_mph_per_play_role[(max_mph_per_play_role['punt_or_return']=='punt')&(max_mph_per_play_role['role_group']!='role_punt_punter')]\n    pivot = filtered.groupby(['role_group',role_group])['max_mph'].mean().to_frame().reset_index().pivot(index='role_group',columns=role_group,values='max_mph')\n    ax = plt.axes()\n    sns.heatmap(pivot, annot=True, fmt='.1f', ax = ax, cmap='Greens')\n    ax.set_title('Impact of %s on Max MPH'%(role_group))\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"95f0fac581d99d750a8c16d200366f13d1a5d900"},"cell_type":"markdown","source":"### Impact on punt team distance by changing personel"},{"metadata":{"trusted":true,"_uuid":"376bf4903b14aa85bba217c5ffc749e2fe6097b9"},"cell_type":"code","source":"# show impact on punt team max_distance by changing a return team position group\nfor role_group in ['role_return_line','role_return_linebacker','role_return_lead_blocker','role_return_jammer']:\n    filtered = avg_distance_per_play_role[(avg_distance_per_play_role['punt_or_return']=='punt')&(avg_distance_per_play_role['role_group']!='role_punt_punter')]\n    pivot = filtered.groupby(['role_group',role_group])['total_distance'].mean().to_frame().reset_index().pivot(index='role_group',columns=role_group,values='total_distance')\n    ax = plt.axes()\n    sns.heatmap(pivot, annot=True, fmt='.1f', ax = ax, cmap='Blues')\n    ax.set_title('Impact of %s on Avg Distance'%(role_group))\n    plt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}