{"cells":[{"metadata":{},"cell_type":"markdown","source":"Below are the analysis for this issue.\nMore details and discussions are provided in my attached slides."},{"metadata":{},"cell_type":"markdown","source":"# Goal: Investigate the relationship between the playing surface and injury and performance of NFL athletes.\n"},{"metadata":{},"cell_type":"markdown","source":"## We analyze this problem from two viewpoints: \n* Basic EDA\n* Model building and model interpretation"},{"metadata":{},"cell_type":"markdown","source":"## The outline of our study is:\n* Data pre-processing\n    * Data cleaning and merge\n    * Data selection and extraction\n* Exploratory data analysis\n    * Insights\n* Model building\n    * Feature engineering\n    * Model building\n    * Model evaluation\n    * Model interpretation\n* Conclusion"},{"metadata":{},"cell_type":"markdown","source":"### Data pre-processing\n#### Data cleaning"},{"metadata":{"_kg_hide-input":false,"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pandas as pd\nimport numpy as np\nimport random\nimport pickle\nimport gc\nfrom matplotlib import pyplot as plt\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import accuracy_score, f1_score, r2_score, roc_auc_score\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load data\nplaylist_df = pd.read_csv('../input/nfl-playing-surface-analytics/PlayList.csv')\nplayerTrackData_df = pd.read_csv('../input/nfl-playing-surface-analytics/PlayerTrackData.csv')\ninjuryRecord_df = pd.read_csv('../input/nfl-playing-surface-analytics/InjuryRecord.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"# Add column 'PlayerKey', 'GameID' and 'PlayerGamePlay'\nplayerTrackData_df['PlayerKey'] = playerTrackData_df['PlayKey'].apply(lambda x: int(x.split('-')[0]))\nplayerTrackData_df['GameID'] = playerTrackData_df['PlayKey'].apply(lambda x: x.split('-')[0] + '-' + x.split('-')[1])\nplayerTrackData_df['PlayerGamePlay'] = playerTrackData_df['PlayKey'].apply(lambda x: int(x.split('-')[2]))\nplayerTrackData_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"# Remove unreasonable values in column 'Temperature'\ntmp_list = list(playlist_df['Temperature'].values)\ntmp_list = [ele for ele in tmp_list if not ele == -999]\nmean_tmp = np.mean(tmp_list)\nplaylist_df['Temperature'] = playlist_df['Temperature'].replace(-999, mean_tmp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Remap values in column 'StadiumType'\noutdoor_str_list = ['Outdoor', 'Outdoors', 'Oudoor', 'Ourdoor', 'Outdoor Retr Roof-Open', 'Outddors', 'Outdor',\n                   'Heinz Field', 'Outside', 'Cloudy']\nplaylist_df['StadiumType'] = playlist_df['StadiumType'].apply(lambda x: 'Outdoor' if x in outdoor_str_list else 'Indoor')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Remap values in column 'Weather'\ncloudy_list = ['Cloudy', 'Partly Cloudy', 'Mostly Cloudy', 'Mostly cloudy', 'Partly cloudy', 'cloudy',\n              'Mostly Coudy', 'Cloudy, chance of rain', 'Partly Clouidy', 'Coudy', 'Party Cloudy'] \nsunny_list = ['Sunny', 'Mostly Sunny', 'Partly Sunny', 'Mostly sunny', 'Sunny and clear', 'Sunny and warm',\n             'Sunny, highs to upper 80s', 'Sunny Skies', 'Clear and Sunny', 'Partly sunny', 'Clear and sunny',\n             'Sunny, Windy', 'Heat Index 95', 'Mostly Sunny Skies', 'Sun & clouds',]\nclear_list = ['Clear', 'Fair', 'Clear and warm', 'Clear Skies', 'Clear skies', 'Partly clear', 'Clear to Partly Cloudy',]\ncold_list = ['Cold', 'Clear and cold', 'Clear and Cool', 'Cloudy and Cool', 'Sunny and cold', ]\nrain_list = ['Rain', 'Light Rain', 'Rain Chance 40%', 'Cloudy, 50% change of rain', 'Cloudy and cold',\n            'Scattered Showers', 'Rain likely, temps in low 40s.', 'Showers', 'Rainy', 'Rain shower', 'Cloudy, Rain',\n            '10% Chance of Rain', '30% Chance of Rain', 'Cloudy with periods of rain, thunder possible. Winds shifting to WNW, 10-20 mph.',]\nhazy_list = ['Hazy', 'Overcast', 'Cloudy, fog started developing in 2nd quarter', ]\nsnow_list = ['Snow', 'Cloudy, light snow accumulating 1-3\"', 'Heavy lake effect snow',]\nindoor_list = ['Controlled Climate', 'N/A (Indoors)', 'Indoors', 'Indoor', 'N/A Indoor',]\n\ndef weather_process(weather_str):\n    if weather_str in cloudy_list:\n        return 'Cloudy'\n    elif weather_str in sunny_list:\n        return 'Sunny'\n    elif weather_str in clear_list:\n        return 'Clear'\n    elif weather_str in cold_list:\n        return 'Cold'\n    elif weather_str in rain_list:\n        return 'Rainy'\n    elif weather_str in hazy_list:\n        return 'Hazy'\n    elif weather_str in snow_list:\n        return 'Snowy'\n    elif weather_str in indoor_list:\n        return 'N/A (Indoor)'\n    else:\n        return 'nan'\nplaylist_df['Weather'] = playlist_df['Weather'].apply(lambda x: weather_process(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Convert 'DM_M1', 'DM_M7', 'DM_M28', 'DM_M42' to column 'seriosity'\ninjuryRecord_df['seriosity'] = injuryRecord_df['DM_M1'] + injuryRecord_df['DM_M7'] + injuryRecord_df['DM_M28'] + injuryRecord_df['DM_M42']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Data selection"},{"metadata":{"trusted":true},"cell_type":"code","source":"inj_PlayerKey_list = list(injuryRecord_df.PlayerKey.values)\ntarget_track_df = playerTrackData_df[playerTrackData_df['PlayerKey'].isin(inj_PlayerKey_list)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"maxPlayerKey_series = target_track_df.groupby('GameID')['PlayerGamePlay'].max()\nmax_play_id_dict = dict(zip(maxPlayerKey_series.index, maxPlayerKey_series.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"injuryRecord_df['PlayKey'] = injuryRecord_df['PlayKey'].astype(str)\ninjuryRecord_df['PlayKey'] = injuryRecord_df.apply(lambda row: row['GameID']+'-'+str(max_play_id_dict[row['GameID']]) if row['PlayKey'] == 'nan' else row['PlayKey'], axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"body_part_dict = dict(zip(injuryRecord_df.PlayKey, injuryRecord_df.BodyPart))\nsurface_dict = dict(zip(injuryRecord_df.PlayKey, injuryRecord_df.Surface))\nseriosity_dict = dict(zip(injuryRecord_df.PlayKey, injuryRecord_df.seriosity))\n\ntarget_track_df['BodyPart'] = target_track_df['PlayKey'].apply(lambda x: body_part_dict[x] if x in body_part_dict else None)\ntarget_track_df['Surface'] = target_track_df['PlayKey'].apply(lambda x: surface_dict[x] if x in surface_dict else None)\ntarget_track_df['Seriosity'] = target_track_df['PlayKey'].apply(lambda x: seriosity_dict[x] if x in seriosity_dict else None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"RosterPosition_dict = dict(zip(playlist_df.PlayKey, playlist_df.RosterPosition))\nPlayerDay_dict = dict(zip(playlist_df.PlayKey, playlist_df.PlayerDay))\nStadiumType_dict = dict(zip(playlist_df.PlayKey, playlist_df.StadiumType))\nFieldType_dict = dict(zip(playlist_df.PlayKey, playlist_df.FieldType))\nTemperature_dict = dict(zip(playlist_df.PlayKey, playlist_df.Temperature))\nWeather_dict = dict(zip(playlist_df.PlayKey, playlist_df.Weather))\nPlayType_dict = dict(zip(playlist_df.PlayKey, playlist_df.PlayType))\nPosition_dict = dict(zip(playlist_df.PlayKey, playlist_df.Position))\nPositionGroup_dict = dict(zip(playlist_df.PlayKey, playlist_df.PositionGroup))\n\ntarget_track_df['RosterPosition'] = target_track_df['PlayKey'].apply(lambda x: RosterPosition_dict[x])\ntarget_track_df['PlayerDay'] = target_track_df['PlayKey'].apply(lambda x: PlayerDay_dict[x])\ntarget_track_df['StadiumType'] = target_track_df['PlayKey'].apply(lambda x: StadiumType_dict[x])\ntarget_track_df['FieldType'] = target_track_df['PlayKey'].apply(lambda x: FieldType_dict[x])\ntarget_track_df['Temperature'] = target_track_df['PlayKey'].apply(lambda x: Temperature_dict[x])\ntarget_track_df['Weather'] = target_track_df['PlayKey'].apply(lambda x: Weather_dict[x])\ntarget_track_df['PlayType'] = target_track_df['PlayKey'].apply(lambda x: PlayType_dict[x])\ntarget_track_df['Position'] = target_track_df['PlayKey'].apply(lambda x: Position_dict[x])\ntarget_track_df['PositionGroup'] = target_track_df['PlayKey'].apply(lambda x: PositionGroup_dict[x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_track_df['dis_diff'] = target_track_df.groupby('PlayKey')['dis'].diff()\ntarget_track_df['s_diff'] = target_track_df.groupby('PlayKey')['s'].diff()\ntarget_track_df['dir_diff'] = target_track_df.groupby('PlayKey')['dir'].diff()\ntarget_track_df['o_diff'] = target_track_df.groupby('PlayKey')['o'].diff()\ntarget_track_df['dis_diff_3'] = target_track_df.groupby('PlayKey')['dis'].diff(3)\ntarget_track_df['s_diff_3'] = target_track_df.groupby('PlayKey')['s'].diff(3)\ntarget_track_df['dir_diff_3'] = target_track_df.groupby('PlayKey')['dir'].diff(3)\ntarget_track_df['o_diff_3'] = target_track_df.groupby('PlayKey')['o'].diff(3)\ntarget_track_df['dis_diff_5'] = target_track_df.groupby('PlayKey')['dis'].diff(5)\ntarget_track_df['s_diff_5'] = target_track_df.groupby('PlayKey')['s'].diff(5)\ntarget_track_df['dir_diff_5'] = target_track_df.groupby('PlayKey')['dir'].diff(5)\ntarget_track_df['o_diff_5'] = target_track_df.groupby('PlayKey')['o'].diff(5)\n\ntarget_track_df['velocity'] = target_track_df['dis_diff']/0.1 \ntarget_track_df['acceleration'] = target_track_df['s_diff']/0.1\ntarget_track_df['dir_speed'] = target_track_df['dir_diff']/0.1 \ntarget_track_df['o_speed'] = target_track_df['o_diff']/0.1 \n\ntarget_track_df['velocity_3'] = target_track_df['dis_diff_3']/0.1 \ntarget_track_df['acceleration_3'] = target_track_df['s_diff_3']/0.1\ntarget_track_df['dir_speed_3'] = target_track_df['dir_diff_3']/0.1 \ntarget_track_df['o_speed_3'] = target_track_df['o_diff_3']/0.1 \n\ntarget_track_df['velocity_5'] = target_track_df['dis_diff_5']/0.1 \ntarget_track_df['acceleration_5'] = target_track_df['s_diff_5']/0.1\ntarget_track_df['dir_speed_5'] = target_track_df['dir_diff_5']/0.1 \ntarget_track_df['o_speed_5'] = target_track_df['o_diff_5']/0.1 \n\ntarget_track_df['angle'] = target_track_df['o'] - target_track_df['dir']\ntarget_track_df['speed'] = target_track_df['dis'] / target_track_df['time']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_track_df = target_track_df.fillna({'speed':0,'acceleration':0, 'dir_speed': 0, 'o_speed':0, \n                                          'velocity_3':0,'acceleration_3':0, 'dir_speed_3': 0, 'o_speed_3':0, \n                                          'velocity_5':0,'acceleration_5':0, 'dir_speed_5': 0, 'o_speed_5':0, \n                                          'angle':0, 'velocity':0})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_track_df[['PlayKey','time', 'dir', 'dis', 'o', 's', 'speed', 'acceleration',\n       'dir_speed', 'o_speed', 'velocity', 'angle','velocity_3','acceleration_3', 'dir_speed_3', 'o_speed_3', 'velocity_5','acceleration_5', 'dir_speed_5', 'o_speed_5']] = target_track_df[['PlayKey','time', 'dir', 'dis', 'o', 's', 'speed', 'acceleration',\n       'dir_speed', 'o_speed', 'velocity', 'angle','velocity_3','acceleration_3', 'dir_speed_3', 'o_speed_3', 'velocity_5','acceleration_5', 'dir_speed_5', 'o_speed_5']].replace([np.inf, -np.inf], 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"inj_playkey_list = list(injuryRecord_df.PlayKey.values)\ntarget_track_df['if_injury'] = target_track_df['PlayKey'].apply(lambda x: 1 if x in inj_playkey_list else 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_track_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"grouped_max = target_track_df[['PlayKey','time', 'dir', 'dis', 'o', 's', 'speed', 'acceleration',\n       'dir_speed', 'o_speed', 'velocity', 'angle','velocity_3','acceleration_3', 'dir_speed_3', 'o_speed_3', 'velocity_5','acceleration_5', 'dir_speed_5', 'o_speed_5']].groupby(by=['PlayKey']).max()\ngrouped_average = target_track_df[['PlayKey','time', 'dir', 'dis', 'o', 's', 'speed', 'acceleration',\n       'dir_speed', 'o_speed', 'velocity', 'angle','velocity_3','acceleration_3', 'dir_speed_3', 'o_speed_3', 'velocity_5','acceleration_5', 'dir_speed_5', 'o_speed_5']].groupby(by=['PlayKey']).mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"markdown","source":"#### Data merge"},{"metadata":{"trusted":true},"cell_type":"code","source":"final_df = pd.DataFrame()\nfinal_df['PlayKey'] = target_track_df.PlayKey.value_counts().index\n\nfinal_df = final_df.merge(grouped_max.reset_index(), on=['PlayKey'])\nfinal_df = final_df.merge(grouped_average.reset_index(), on=['PlayKey'], suffixes=('_max', '_avg'))\n\ninj_playkey_list = list(injuryRecord_df.PlayKey.values)\nfinal_df['if_injury'] = final_df['PlayKey'].apply(lambda x: 1 if x in inj_playkey_list else 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"RosterPosition_dict = dict(zip(playlist_df.PlayKey, playlist_df.RosterPosition))\nPlayerDay_dict = dict(zip(playlist_df.PlayKey, playlist_df.PlayerDay))\nStadiumType_dict = dict(zip(playlist_df.PlayKey, playlist_df.StadiumType))\nFieldType_dict = dict(zip(playlist_df.PlayKey, playlist_df.FieldType))\nTemperature_dict = dict(zip(playlist_df.PlayKey, playlist_df.Temperature))\nWeather_dict = dict(zip(playlist_df.PlayKey, playlist_df.Weather))\nPlayType_dict = dict(zip(playlist_df.PlayKey, playlist_df.PlayType))\nPosition_dict = dict(zip(playlist_df.PlayKey, playlist_df.Position))\nPositionGroup_dict = dict(zip(playlist_df.PlayKey, playlist_df.PositionGroup))\nPlayerGame_dict = dict(zip(playlist_df.PlayKey, playlist_df.PlayerGame))\nPlayerGamePlay_dict = dict(zip(playlist_df.PlayKey, playlist_df.PlayerGamePlay))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"final_df['RosterPosition'] = final_df['PlayKey'].apply(lambda x: RosterPosition_dict[x])\nfinal_df['PlayerDay'] = final_df['PlayKey'].apply(lambda x: PlayerDay_dict[x])\nfinal_df['StadiumType'] = final_df['PlayKey'].apply(lambda x: StadiumType_dict[x])\nfinal_df['FieldType'] = final_df['PlayKey'].apply(lambda x: FieldType_dict[x])\nfinal_df['Temperature'] = final_df['PlayKey'].apply(lambda x: Temperature_dict[x])\nfinal_df['Weather'] = final_df['PlayKey'].apply(lambda x: Weather_dict[x])\nfinal_df['PlayType'] = final_df['PlayKey'].apply(lambda x: PlayType_dict[x])\nfinal_df['Position'] = final_df['PlayKey'].apply(lambda x: Position_dict[x])\nfinal_df['PositionGroup'] = final_df['PlayKey'].apply(lambda x: PositionGroup_dict[x])\nfinal_df['PlayerGame'] = final_df['PlayKey'].apply(lambda x: PlayerGame_dict[x])\nfinal_df['PlayerGamePlay'] = final_df['PlayKey'].apply(lambda x: PlayerGamePlay_dict[x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"body_part_dict = dict(zip(injuryRecord_df.PlayKey, injuryRecord_df.BodyPart))\nsurface_dict = dict(zip(injuryRecord_df.PlayKey, injuryRecord_df.Surface))\nseriosity_dict = dict(zip(injuryRecord_df.PlayKey, injuryRecord_df.seriosity))\nfinal_df['BodyPart'] = final_df['PlayKey'].apply(lambda x: body_part_dict[x] if x in body_part_dict else 'nan')\nfinal_df['Seriosity'] = final_df['PlayKey'].apply(lambda x: seriosity_dict[x]if x in seriosity_dict else None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"final_df['PlayerGame'] = final_df['PlayKey'].apply(lambda x: int(x.split('-')[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"final_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"final_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"final_df['if_injury'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\n\nplt.style.use(\"ggplot\")\nplt.figure(figsize=(12,8))\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\n\n# seaborn histogram\nsns.distplot(np.array(final_df[final_df['if_injury'] == 0]['o_speed_max']), hist=False, kde=True, kde_kws=dict(linewidth=2),\n             bins=int(50), color = '#1f77b4', label=\"Non injury\")\n\nsns.distplot(np.array(final_df[final_df['if_injury'] == 1]['o_speed_max']), hist=False, kde=True, \n             bins=int(50), color = '#d62728', label=\"Injury: total\", kde_kws=dict(linewidth=2))\n\nplt.xlabel('Distribution of maximum speed per play (yard/sec)', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\nplt.legend(loc=\"upper right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\n\nplt.style.use(\"ggplot\")\nplt.figure(figsize=(12,8))\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\n\nmetric = 'speed_max'\nsns.distplot(np.array(final_df[final_df['if_injury'] == 1][metric]), hist=False, kde=True, \n             bins=int(50), color = '#d62728', label=\"Injury: Total\", kde_kws=dict(linewidth=3))\n\n# seaborn histogram\nsns.distplot(np.array(final_df[(final_df['if_injury'] == 1)&(final_df['BodyPart']=='Knee')][metric]), hist=False, kde=True, \n             bins=int(50),  color = 'green', label=\"Injury: Knee\", kde_kws=dict(linewidth=3))\n\nsns.distplot(np.array(final_df[(final_df['if_injury'] == 1)&(final_df['BodyPart']=='Ankle')][metric]), hist=False, kde=True, \n             bins=int(50),  color = 'orange', label=\"Injury: Ankle\", kde_kws=dict(linewidth=3))\n\nsns.distplot(np.array(final_df[(final_df['if_injury'] == 1)&(final_df['BodyPart']=='Toes')][metric]), hist=False, kde=True, \n             bins=int(50),  color = 'steelblue', label=\"Injury: Toes\", kde_kws=dict(linewidth=3))\n\nsns.distplot(np.array(final_df[(final_df['if_injury'] == 1)&(final_df['BodyPart']=='Foot')][metric]), hist=False, kde=True, \n             bins=int(50),  color = 'black', label=\"Injury: Foot\", kde_kws=dict(linewidth=3))\n\nsns.distplot(np.array(final_df[(final_df['if_injury'] == 1)&(final_df['BodyPart']=='Heel')][metric]), hist=True, kde=False, \n             bins=int(5),  color = 'gray', label=\"Injury: Heel\")\n\nplt.xlabel('Distribution of maximum speed per play (yard/sec)', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\nplt.legend(loc=\"upper right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### TF-IDF Model"},{"metadata":{"trusted":false},"cell_type":"code","source":"target_track_df['event'] = target_track_df['event'].astype(str)\nevent_list = target_track_df.groupby('PlayKey')['event'].apply(list)\nsession_dict = dict(zip(event_list.index, event_list.values))\nplaykey_session_df = pd.DataFrame()\nplaykey_session_df['PlayKey'] = pd.Series(list(session_dict.keys())).values\nplaykey_session_df['session'] = pd.Series(list(session_dict.values())).values\nplaykey_session_df['session'] = playkey_session_df['session'].apply(lambda x : list(filter(('nan').__ne__, x)))\nplaykey_session_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\n\nmlb = MultiLabelBinarizer()\nplaykey_session_df_oh = playkey_session_df.join(pd.DataFrame(mlb.fit_transform(playkey_session_df.pop('session')),\n                          columns=mlb.classes_,\n                          index=playkey_session_df.index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"playkey_session_df_oh.loc['Total',:]= playkey_session_df_oh.sum(axis=0)\nwordFreqDict = playkey_session_df_oh.tail(1).to_dict('records')[0]\ndel wordFreqDict['PlayKey']","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"wordFreq_df = pd.DataFrame()\nwordFreq_df['word'] = pd.Series(list(wordFreqDict.keys())).values\nwordFreq_df['freq'] = pd.Series(list(wordFreqDict.values())).values\nwordFreq_df = wordFreq_df.sort_values('freq', ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\n\nplt.figure(figsize = (20, 6))\nword_freq_plot = sns.barplot(data = wordFreq_df, x = \"word\", y = \"freq\")\nword_freq_plot.set_xticklabels(word_freq_plot.get_xticklabels(), rotation=90, fontsize=15)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"def remove_kickoff_snap(input_list):\n    if 'kickoff' in input_list:\n        return input_list[input_list.index('kickoff')+1:]\n    elif 'ball_snap' in input_list:\n        return input_list[input_list.index('ball_snap')+1:]\n    else:\n        return input_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"raw_corpus = playkey_session_df.session.values\ntotal_corpus = []\nfor single_list in raw_corpus:\n    sent = ' '.join(single_list)\n    total_corpus.append(sent)\nlen(total_corpus)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nvectorizer = CountVectorizer()\nX = vectorizer.fit_transform(total_corpus) \nword = vectorizer.get_feature_names() \n\nfrom sklearn.feature_extraction.text import TfidfTransformer  \n\ntransformer = TfidfTransformer() \ntfidf = transformer.fit_transform(X)  \nprint (tfidf.toarray().shape)  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"tfidf_df = pd.DataFrame(tfidf.toarray(), columns = word)\ntfidf_df['PlayKey'] = playkey_session_df['PlayKey']","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"playkey_last_event_df = pd.DataFrame()\nplaykey_last_event_df['PlayKey'] = playkey_session_df['PlayKey']\nplaykey_last_event_df['last_event'] = playkey_session_df['session'].apply(lambda x : x[-1:][0])\nlen(playkey_last_event_df['last_event'].value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"train_df = final_df.merge(tfidf_df.reset_index(), on=['PlayKey'], how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"train_df = train_df.merge(playkey_last_event_df.reset_index(), on=['PlayKey'], how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"train_df_syn = train_df[train_df['FieldType'] == 'Synthetic']\ntrain_df_nat = train_df[train_df['FieldType'] == 'Natural']","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"print(list(train_df_syn.columns))","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"# dense_cols = ['time_max','s_max', 'speed_max', 'acceleration_max', 'dir_speed_max', 'o_speed_max', 'velocity_max', 'angle_max',\n#               'dir_avg', 'dis_avg', 'o_avg', 's_avg', 'speed_avg', 'acceleration_avg', 'dir_speed_avg', 'o_speed_avg', \n#               'velocity_avg', 'angle_avg', 'PlayerDay', 'Temperature', \n#              'ball_snap', 'end_path', 'extra_point', 'extra_point_attempt', 'extra_point_blocked', 'extra_point_fake', 'extra_point_missed',\n#             'fair_catch', 'field_goal', 'field_goal_attempt', 'field_goal_blocked', 'field_goal_fake', 'field_goal_missed', 'field_goal_play', \n#             'first_contact', 'free_kick', 'free_kick_play', 'fumble', 'fumble_defense_recovered', 'fumble_offense_recovered', 'handoff', \n#             'huddle_break_offense', 'huddle_start_offense', 'kick_received', 'kick_recovered', 'kickoff', 'kickoff_land', 'kickoff_play',\n#             'lateral', 'line_set', 'man_in_motion', 'onside_kick', 'out_of_bounds', 'pass_arrived', 'pass_forward', 'pass_lateral', \n#             'pass_outcome_caught', 'pass_outcome_incomplete', 'pass_outcome_interception', 'pass_outcome_touchdown', 'pass_shovel',\n#             'pass_tipped', 'penalty_accepted', 'penalty_declined', 'penalty_flag', 'play_action', 'play_submit', 'punt', 'punt_blocked',\n#             'punt_downed', 'punt_fake', 'punt_land', 'punt_muffed', 'punt_play', 'punt_received', 'qb_kneel', 'qb_sack', 'qb_spike', \n#             'qb_strip_sack', 'run', 'run_pass_option', 'safety', 'shift', 'snap_direct', 'tackle', 'timeout', 'timeout_away', \n#             'timeout_booth_review', 'timeout_halftime', 'timeout_home', 'timeout_injury', 'timeout_quarter', 'timeout_tv', 'touchback',\n#             'touchdown', 'two_minute_warning', 'two_point_conversion', 'two_point_play', 'xp_fake'\n#              ]\n\n# dense_cols = ['time_max','s_max', 'speed_max', 'acceleration_max', 'dir_speed_max', 'o_speed_max', 'velocity_max', 'angle_max',\n#               'dir_avg', 'dis_avg', 'o_avg', 's_avg', 'speed_avg', 'acceleration_avg', 'dir_speed_avg', 'o_speed_avg', \n#               'velocity_avg', 'angle_avg', 'PlayerDay', 'Temperature', 'PlayerGame'\n             \n#              ]\n\ndense_cols = ['time_max','s_max', 'speed_max', 'acceleration_max', 'dir_speed_max', 'o_speed_max', 'velocity_max','angle_max',\n              'velocity_5_max','acceleration_5_max', 'dir_speed_5_max', 'o_speed_5_max', \n              'velocity_5_avg','acceleration_5_avg', 'dir_speed_5_avg', 'o_speed_5_avg', \n              'velocity_3_max','acceleration_3_max', 'dir_speed_3_max', 'o_speed_3_max', \n              'velocity_3_avg','acceleration_3_avg', 'dir_speed_3_avg', 'o_speed_3_avg',\n              'dir_avg', 'dis_avg', 'o_avg', 's_avg', 'speed_avg', 'acceleration_avg', 'dir_speed_avg', 'o_speed_avg', \n              'velocity_avg', 'angle_avg', 'PlayerDay', 'Temperature', \n             'ball_snap', 'end_path', 'extra_point', 'extra_point_attempt', 'extra_point_blocked', 'extra_point_fake', 'extra_point_missed',\n            'fair_catch', 'field_goal', 'field_goal_attempt', 'field_goal_blocked', 'field_goal_fake', 'field_goal_missed', 'field_goal_play', \n            'first_contact', 'free_kick', 'free_kick_play', 'fumble', 'fumble_defense_recovered', 'fumble_offense_recovered', 'handoff', \n            'huddle_break_offense', 'huddle_start_offense', 'kick_received', 'kick_recovered', 'kickoff', 'kickoff_land', 'kickoff_play',\n            'lateral', 'line_set', 'man_in_motion', 'onside_kick', 'out_of_bounds', 'pass_arrived', 'pass_forward', 'pass_lateral', \n            'pass_outcome_caught', 'pass_outcome_incomplete', 'pass_outcome_interception', 'pass_outcome_touchdown', 'pass_shovel',\n            'pass_tipped', 'penalty_accepted', 'penalty_declined', 'penalty_flag', 'play_action', 'play_submit', 'punt', 'punt_blocked',\n            'punt_downed', 'punt_fake', 'punt_land', 'punt_muffed', 'punt_play', 'punt_received', 'qb_kneel', 'qb_sack', 'qb_spike', \n            'qb_strip_sack', 'run', 'run_pass_option', 'safety', 'shift', 'snap_direct', 'tackle', 'timeout', 'timeout_away', \n            'timeout_booth_review', 'timeout_halftime', 'timeout_home', 'timeout_injury', 'timeout_quarter', 'timeout_tv', 'touchback',\n            'touchdown', 'two_minute_warning', 'two_point_conversion', 'two_point_play', 'xp_fake', 'PlayerGame'\n             ]\n\nlen(dense_cols)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"cate_cols = ['RosterPosition', 'StadiumType', 'FieldType', 'Weather', 'PlayType', 'Position', 'last_event']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Synthetic"},{"metadata":{"trusted":false},"cell_type":"code","source":"train_df_syn = train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"y_syn = train_df_syn['if_injury']","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"train_df_syn['if_injury'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"cate_feat_syn = pd.get_dummies(train_df_syn[cate_cols])\n\ndense_feat_syn = train_df_syn[dense_cols]\ndense_feat_syn = dense_feat_syn.replace([np.inf, -np.inf], 0)\nfrom sklearn.preprocessing import RobustScaler\nscaler = RobustScaler()\ndense_feat_syn = scaler.fit_transform(dense_feat_syn.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"cate_feat_arr_syn = cate_feat_syn.to_numpy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"final_train_arr_syn = np.concatenate((cate_feat_arr_syn, dense_feat_syn), axis=1)\nfinal_train_arr_syn.shape, y_syn.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"weight_ratio = float(list(y_syn).count(1)/list(y_syn).count(0))\nw_array = np.array([1]*y_syn.shape[0])\nw_array = [1- weight_ratio if ele == 1 else weight_ratio for ele in y_syn]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"from xgboost import XGBClassifier\nxgbc_syn = XGBClassifier()\nxgbc_syn.fit(final_train_arr_syn, y_syn, sample_weight=w_array)\nprint(xgbc_syn.score(final_train_arr_syn, y_syn))\ny_pred_proba_syn = xgbc_syn.predict_proba(final_train_arr_syn)\ny_pred_syn = xgbc_syn.predict(final_train_arr_syn)\nconf_mat_syn = confusion_matrix(y_syn, y_pred_syn)\nprint(conf_mat_syn)\nprint(classification_report(y_syn, y_pred_syn))","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, precision_recall_curve\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nconf_matrix = confusion_matrix(y_syn, y_pred_syn)\nplt.figure(figsize=(8, 8))\nsns.set(font_scale = 2.5)\ng = sns.heatmap(conf_matrix, xticklabels=['Non-fastball', 'Fastball'], yticklabels=['Non-fastball', 'Fastball'], annot=True, fmt=\"d\");\nplt.title(\"Confusion matrix\")\nplt.ylabel('True class', fontsize=20)\nplt.xlabel('Predicted class', fontsize=20)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"y_pred_proba_pos_syn = [ele[1] for ele in y_pred_proba_syn]\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.font_manager import FontProperties\n\nimport seaborn as sns\n\n# Figures inline and set visualization style\n%matplotlib inline\n\nmyfont = FontProperties(fname='/System/Library/Fonts/STHeiti Medium.ttc', size = 14)\nsns.set(font=myfont.get_name())\n\nfrom sklearn.metrics import recall_score, classification_report, auc, roc_curve\nfrom sklearn.metrics import precision_recall_fscore_support, f1_score\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport numpy as np\nplt.gcf().set_size_inches(10, 5)\nsns.set(font_scale = 1.5)\n\nfalse_pos_rate, true_pos_rate, thresholds = roc_curve(y_syn, y_pred_proba_pos_syn)\n\nroc_auc = auc(false_pos_rate, true_pos_rate,)\n\nplt.plot(false_pos_rate, true_pos_rate, linewidth=3, label='XGBoost (%0.3f)'% roc_auc)\n\nplt.plot([0,1],[0,1], linewidth=3)\n\nplt.xlim([-0.01, 1])\nplt.ylim([0, 1.01])\nplt.legend(loc='lower right')\nplt.title('ROC curve')\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"feature_importance_dict = dict()\nimportance_arr = xgbc_syn.feature_importances_\nfor i, col in enumerate(list(cate_feat_syn.columns)+dense_cols):\n    feature_importance_dict[col] = importance_arr[i]\n\nimportance_df = pd.DataFrame()\nimportance_df['feat'] = pd.Series(list(feature_importance_dict.keys())).values\nimportance_df['importance'] = pd.Series(list(feature_importance_dict.values())).values\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\n\nplt.figure(figsize = (6, 10))\nsns.barplot(data = importance_df.sort_values(by = \"importance\", ascending = False).head(35), x = \"importance\", y = \"feat\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"X_shap_syn = pd.DataFrame(final_train_arr_syn, columns=list(cate_feat_syn.columns)+dense_cols)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"import shap\n\n# load JS visualization code to notebook\nshap.initjs()\n\n# explain the model's predictions using SHAP\n# (same syntax works for LightGBM, CatBoost, scikit-learn and spark models)\nexplainer = shap.TreeExplainer(xgbc_syn)\nshap_values = explainer.shap_values(X_shap_syn)\n\n# visualize the first prediction's explanation (use matplotlib=True to avoid Javascript)\nshap.force_plot(explainer.expected_value, shap_values[0,:], X_shap_syn.iloc[0,:])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"shap.summary_plot(shap_values, X_shap_syn, max_display=30)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":false},"cell_type":"code","source":"shap.summary_plot(shap_values, X_shap_syn, plot_type=\"bar\", max_display=50)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"target_track_df['if_injury'] = target_track_df['PlayKey'].apply(lambda x: 1 if x in inj_playkey_list else 0)\ntarget_track_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"## import matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.misc import imread\n\ndf = target_track_df[target_track_df['if_injury']==1]\nplaykey_list = list(df['PlayKey'].value_counts().index)\n\nsingle_playkey = playkey_list[2]\ntemp_df = df[df['PlayKey']==single_playkey]\ntemp_df['event'] = temp_df['event'].astype(str)\n\nplt.style.use('ggplot')\nplt.figure(figsize=(25,10))\n# img = imread('./pitch.png')\n# plt.imshow(img, zorder=1, extent=[0.0, 120, 0.0, 53.3])\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\nplt.xlabel('Rank', fontsize=20)\nplt.ylabel('Euclidean distance', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\n# plt.xticks(x, rank)\nfor i, single_playkey in enumerate(playkey_list[:1]):\n    plt.plot(temp_df['x'].values, temp_df['y'].values, '-o', linewidth=3, markersize=7, color='steelblue', markeredgewidth=0.0)\n    \nevnt_list = [ele for ele in list(temp_df['event']) if ele!='nan'] \ntime_counter = -1\nfor evnt in evnt_list:\n    cnt = 0\n    while( temp_df[temp_df['event'] == evnt]['time'].values[cnt] <= time_counter):\n        cnt+=1\n    time_counter = temp_df[temp_df['event'] == evnt]['time'].values[cnt]\n    x = temp_df[temp_df['event'] == evnt]['x'].values[cnt]\n    y = temp_df[temp_df['event'] == evnt]['y'].values[cnt]\n    plt.text(x, y+0.3, evnt, rotation=90, fontsize = 20)\n    plt.arrow(temp_df[temp_df['event'] == evnt]['x'].values[cnt], temp_df[temp_df['event'] == evnt]['y'].values[cnt], 0.3*np.cos(temp_df[temp_df['event'] == evnt]['dir'].values[cnt]*np.pi/180), \n              0.3*np.sin(temp_df[temp_df['event'] == evnt]['dir'].values[cnt]*np.pi/180), head_width=0.1, head_length=0.2, color='red')\n    plt.arrow(temp_df[temp_df['event'] == evnt]['x'].values[cnt], temp_df[temp_df['event'] == evnt]['y'].values[cnt], 0.3*np.cos(temp_df[temp_df['event'] == evnt]['o'].values[cnt]*np.pi/180), \n              0.3*np.sin(temp_df[temp_df['event'] == evnt]['o'].values[cnt]*np.pi/180), head_width=0.1, head_length=0.2, color='green')\n\naxes = plt.gca()\naxes.set_xlim([12, 28])\naxes.set_ylim([22, 30])\nplt.legend(loc=\"lower right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"print(temp_df['dir'][20721811])\nnp.cos(temp_df['dir'][20721811]*np.pi/180)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"df[df['PlayKey']==single_playkey]['x'].values[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"df = target_track_df[target_track_df['if_injury']==1]\nplaykey_list = list(df['PlayKey'].value_counts().index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"## import matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.misc import imread\n\n# df = target_track_df[target_track_df['if_injury']==0]\n# playkey_list = list(df['PlayKey'].value_counts().index)\n\nplt.style.use('ggplot')\nplt.figure(figsize=(12,8))\nimg = imread('./pitch.png')\nplt.imshow(img, zorder=1, extent=[0.0, 120, 0.0, 53.3])\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\nplt.xlabel('Rank', fontsize=20)\nplt.ylabel('Euclidean distance', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\n# plt.xticks(x, rank)\nfor i, single_playkey in enumerate(playkey_list[:100]):\n    plt.plot(df[df['PlayKey']==single_playkey]['x'].values, df[df['PlayKey']==single_playkey]['y'].values, '-o', linewidth=2, markersize=5)\nplt.legend(loc=\"lower right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"numOfPlayers_gameid = list(playlist_df.groupby(['PlayerGame'])['PlayerKey'].nunique())\nnumOfInj_gameid = list(final_df.groupby('PlayerGame')['if_injury'].sum().values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\nplt.style.use(\"ggplot\")\nplt.figure(figsize=(15,8))\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\nplt.xlabel('Rank', fontsize=20)\nplt.ylabel('Euclidean distance', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\nplt.xticks(list(range(0,32)), list(range(1,33)))\nplt.plot(numOfPlayers_gameid, '-o', color='#1f77b4', label=\"Successful rank-1 identification\", linewidth=3, markersize=10)\n# plt.legend(loc=\"lower right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"np.array(numOfInj_gameid)/np.array(numOfPlayers_gameid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\nplt.style.use(\"ggplot\")\nplt.figure(figsize=(15,8))\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\nplt.xlabel('Rank', fontsize=20)\nplt.ylabel('Euclidean distance', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\nplt.xticks(list(range(0,32)), list(range(1,33)))\nplt.plot(np.array(numOfInj_gameid)/np.array(numOfPlayers_gameid)*100, '-o', color='#1f77b4', label=\"Successful rank-1 identification\", linewidth=3, markersize=10)\n#plt.plot(numOfInj_gameid, '-o', color='#d62728', label=\"Failed rank-1 identification\", linewidth=3, markersize=10)\n# plt.legend(loc=\"lower right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"axes[1].axvline(x=temp_df[temp_df['event'] == evnt]['time'].values[0], color = 'r')","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"## field\n## last event\n## gameId -- injury","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"## import matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.misc import imread\n\n# df = target_track_df[target_track_df['if_injury']==0]\n# playkey_list = list(df['PlayKey'].value_counts().index)\nsingle_playkey = playkey_list[2]\ntemp_df = df[df['PlayKey']==single_playkey]\ntemp_df['event'] = temp_df['event'].astype(str)\nplt.style.use('ggplot')\nplt.figure(figsize=(25,8))\n#plt.title('Accuracy of face identification at different ranks', fontsize=20)\nplt.xlabel('Rank', fontsize=20)\nplt.ylabel('Euclidean distance', fontsize=20)\nplt.xticks(fontsize=18)\nplt.yticks(fontsize=18)\n# plt.xticks(x, rank)\n\nplt.plot(df[df['PlayKey']==single_playkey]['time'].values, df[df['PlayKey']==single_playkey]['angle'].values,  '-', linewidth=2, color= 'steelblue', markersize=5)\nevnt_list = [ele for ele in list(temp_df['event']) if ele!='nan'] \ntime_counter = -1\nfor evnt in evnt_list:\n    cnt = 0\n    while( temp_df[temp_df['event'] == evnt]['time'].values[cnt] <= time_counter):\n        cnt+=1\n    plt.axvline(x=temp_df[temp_df['event'] == evnt]['time'].values[cnt], color = 'r')\n    time_counter = temp_df[temp_df['event'] == evnt]['time'].values[cnt]\n    plt.text(temp_df[temp_df['event'] == evnt]['time'].values[cnt]+0.11,0,evnt, rotation=90)\nplt.legend(loc=\"lower right\", fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"evnt_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"temp_df[temp_df['event'] == 'penalty_flag']['time'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"temp_df[['speed','time', 'event']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"df = target_track_df[target_track_df['if_injury']==1]\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"len(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"df.to_csv('inj_track_df.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"df['Weather'].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.4"}},"nbformat":4,"nbformat_minor":1}