{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport itertools\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy.stats as ss\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom imblearn.over_sampling import RandomOverSampler, SMOTE\nimport xgboost as xgb\nfrom sklearn.metrics import accuracy_score, confusion_matrix, cohen_kappa_score","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Load Data\n\nLoad data as dataframes"},{"metadata":{"trusted":true},"cell_type":"code","source":"injury_df = pd.read_csv('/kaggle/input/nfl-playing-surface-analytics/InjuryRecord.csv')\nplay_df = pd.read_csv('/kaggle/input/nfl-playing-surface-analytics/PlayList.csv')\nplayer_df = pd.read_csv('/kaggle/input/nfl-playing-surface-analytics/PlayerTrackData.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### General Information\n\nPlay list contains PlayerKey, GameID, and PlayKey, which are unique identifiers. Let's see the number of unique IDs."},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_players = play_df.PlayerKey.nunique()\nunique_games = play_df.GameID.nunique()\nunique_plays = play_df.PlayKey.nunique()\n\nprint('Number of unique players = %d' % unique_players)\nprint('Number of unique games = %d' % unique_games)\nprint('Number of unique plays = %d' % unique_plays)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Game Exploration"},{"metadata":{"trusted":true},"cell_type":"code","source":"# dataframe with game level information\ngame_df = play_df[['GameID', 'StadiumType', 'FieldType', 'Temperature', 'Weather']]\ngame_df = game_df.drop_duplicates()  # the above will return duplicate entries because they are obtained from plays\ngame_df = game_df.reset_index()  # dropping duplicates will preserve the index and so reset it\ngame_df = game_df.drop(columns=['index'])  # index column is not needed anyways and so drop it","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://stackoverflow.com/questions/28931224/adding-value-labels-on-a-matplotlib-bar-chart\ndef add_value_labels(ax, spacing=5):\n    for rect in ax.patches:\n        y_value = rect.get_height()\n        x_value = rect.get_x() + rect.get_width() / 2\n        \n        space = spacing\n        va = 'bottom'\n        \n        if y_value < 0:\n            space *= -1\n            va = 'top'\n        \n        label = \"{:.0f}\".format(y_value)\n        \n        ax.annotate(\n            label,\n            (x_value, y_value),\n            xytext=(0, space),\n            textcoords=\"offset points\",\n            ha='center',\n            va=va\n        )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_game_features(game_df, rotation=90, add_labels=False, figsize=(10, 10)):\n    plt.style.use('ggplot')\n    fig = plt.figure(figsize=figsize)\n    grid = plt.GridSpec(4, 3, hspace=0.2, wspace=0.2)\n    stadium_ax = fig.add_subplot(grid[0, :2])\n    fieldtype_ax = fig.add_subplot(grid[0, 2])\n    weather_ax = fig.add_subplot(grid[1, 0:])\n    temperature_ax = fig.add_subplot(grid[2, 0:])\n    temperature_box_ax = fig.add_subplot(grid[3, 0:])\n    \n    stadium_ax.bar(\n        game_df.StadiumType.value_counts().keys(),\n        game_df.StadiumType.value_counts().values,\n        color='#00c2c7'\n    )\n    stadium_ax.set_title('StadiumType')\n    stadium_ax.set_xticklabels(\n        game_df.StadiumType.value_counts().keys(),\n        rotation=rotation\n    )\n    if add_labels:\n        add_value_labels(stadium_ax, spacing=5)\n    \n    fieldtype_ax.bar(\n        game_df.FieldType.value_counts().keys(),\n        game_df.FieldType.value_counts().values,\n        color=['#00c2c7', '#ff9e15']\n    )\n    fieldtype_ax.set_title('FieldType')\n    fieldtype_ax.set_xticklabels(\n        game_df.FieldType.value_counts().keys(),\n        rotation=0\n    )\n    if add_labels:\n        add_value_labels(fieldtype_ax, spacing=5)\n    \n    weather_ax.bar(\n        game_df.Weather.value_counts().keys(),\n        game_df.Weather.value_counts().values,\n        color='#00c2c7'\n    )\n    weather_ax.set_title('Weather')\n    weather_ax.set_xticklabels(\n        game_df.Weather.value_counts().keys(),\n        rotation=rotation\n    )\n    if add_labels:\n        add_value_labels(weather_ax, spacing=5)\n    \n    temperature_ax.hist(\n        game_df.Temperature.astype(int).values, \n        bins=30,\n        range=(0, 90)\n    )\n    temperature_ax.set_xlim(0, 110)\n    temperature_ax.set_xticks(range(0, 110, 10))\n    temperature_ax.set_xticklabels(range(0, 110, 10))\n    temperature_ax.set_title('Temperature')\n    \n    temperature_box_ax.boxplot(\n        game_df.Temperature.astype(int).values,\n        vert=False\n    )\n    temperature_box_ax.set_xlim(0, 110)\n    temperature_box_ax.set_xticks(range(0, 110, 10))\n    temperature_box_ax.set_xticklabels(range(0, 110, 10))\n    temperature_box_ax.set_yticklabels(['Temperature'])\n    \n    plt.suptitle('Game-Level Exploration', fontsize=16)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_weather(row):\n    # bucket weather into 5 categories\n    cloudy = [\n        'Cloudy 50% change of rain', 'Hazy', 'Cloudy.', 'Overcast', \n        'Mostly Cloudy', 'Cloudy, fog started developing in 2nd quarter', \n        'Partly Cloudy', 'Mostly cloudy', 'Rain Chance 40%', ' Partly cloudy', 'Party Cloudy',\n        'Rain likely, temps in low 40s', 'Partly Clouidy', 'Cloudy, 50% change of rain', \n        'Mostly Coudy', '10% Chance of Rain', 'Cloudy, chance of rain', \n        '30% Chance of Rain', 'Cloudy, light snow accumulating 1-3\"', 'cloudy', 'Coudy', \n        'Cloudy with periods of rain, thunder possible. Winds shifting to WNW, 10-20 mph.',\n        'Cloudy fog started developing in 2nd quarter', 'Cloudy light snow accumulating 1-3\"',\n        'Cloudywith periods of rain, thunder possible. Winds shifting to WNW, 10-20 mph.',\n        'Cloudy 50% change of rain', 'Cloudy and cold', 'Cloudy and Cool', 'Partly cloudy'\n    ]\n    \n    clear = [\n        'Clear, Windy',' Clear to Cloudy', 'Clear, highs to upper 80s', 'Clear and clear',\n        'Partly sunny', 'Clear, Windy', 'Clear skies', 'Sunny', 'Partly Sunny', 'Mostly Sunny', \n        'Clear Skies', 'Sunny Skies', 'Partly clear', 'Fair', 'Sunny, highs to upper 80s', \n        'Sun & clouds', 'Mostly sunny','Sunny, Windy', 'Mostly Sunny Skies', 'Clear and Sunny', \n        'Clear and sunny','Clear to Partly Cloudy', 'Clear Skies', 'Clear and cold', \n        'Clear and warm', 'Clear and Cool', 'Sunny and cold', 'Sunny and warm', 'Sunny and clear'\n    ]\n    \n    rainy = [\n        'Rainy', 'Scattered Showers', 'Showers', 'Cloudy Rain', 'Light Rain', \n        'Rain shower', 'Rain likely, temps in low 40s.', 'Cloudy, Rain'\n    ]\n    \n    snow = ['Heavy lake effect snow']\n    \n    indoor = ['Controlled Climate', 'Indoors', 'N/A Indoor', 'N/A (Indoors)']\n    \n    if row.Weather in cloudy:\n        return 'Cloudy'\n    if row.Weather in indoor:\n        return 'Indoor'\n    if row.Weather in clear:\n        return 'Clear'\n    if row.Weather in rainy:\n        return 'Rain'\n    if row.Weather in snow:\n        return 'Snow'\n    if row.Weather in ['Cloudy.', 'Heat Index 95', 'Cold']:\n        return np.nan\n    return row.Weather\n\n\ndef clean_stadium_type(row):\n    if row.StadiumType in ['Bowl', 'Heinz Field', 'Cloudy']:\n        return np.nan\n    return row.StadiumType\n\n\ndef clean_play_df(play_df):\n    play_df_cleaned = play_df.copy()\n    \n    # clean StadiumType\n    play_df_cleaned['StadiumType'] = play_df_cleaned['StadiumType'].str.replace(\n        r'Oudoor|Outdoors|Ourdoor|Outddors|Outdor|Outside', \n        'Outdoor'\n    )\n    play_df_cleaned['StadiumType'] = play_df_cleaned['StadiumType'].str.replace(\n        r'Indoors|Indoor, Roof Closed|Indoor, Open Roof', \n        'Indoor'\n    )\n    play_df_cleaned['StadiumType'] = play_df_cleaned['StadiumType'].str.replace(\n        r'Closed Dome|Domed, closed|Domed, Open|Domed, open|Dome, closed|Domed', \n        'Dome'\n    )\n    play_df_cleaned['StadiumType'] = play_df_cleaned['StadiumType'].str.replace(\n        r'Retr. Roof-Closed|Outdoor Retr Roof-Open|Retr. Roof - Closed|Retr. Roof-Open|Retr. Roof - Open|Retr. Roof Closed', \n        'Retractable Roof'\n    )\n    play_df_cleaned['StadiumType'] = play_df_cleaned.apply(lambda row : clean_stadium_type(row), axis=1)\n    \n    # clean weather\n    play_df_cleaned['Weather'] = play_df_cleaned.apply(lambda row : clean_weather(row), axis=1)\n    \n    return play_df_cleaned","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"play_df_cleaned = clean_play_df(play_df)\ngame_df_cleaned = play_df_cleaned[[\n    'GameID', \n    'StadiumType', \n    'FieldType', \n    'Weather', \n    'Temperature'\n]]\ngame_df_cleaned = game_df_cleaned.drop_duplicates()\ngame_df_cleaned = game_df_cleaned.reset_index().drop(columns=['index'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_game_features(\n    game_df_cleaned,\n    rotation=0,\n    add_labels=True,\n    figsize=(12, 16)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"percent_diff_in_turfs = 100 * (3311-2401)/(3311+2401)\nprint(\"Percentage difference in the turfs: {:.0f}\".format(percent_diff_in_turfs))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The above diagrams show:\n* **The majority of the games are played outdoors.** So, weather is very important.\n* Percentage difference between synthetic and natural turfs is very less and is not that useful.\n* **The temperature and other weather conditions vary greately**, which could be an interesting aspect to look at."},{"metadata":{},"cell_type":"markdown","source":"#### Player level exploration"},{"metadata":{"trusted":true},"cell_type":"code","source":"player_data_df = play_df_cleaned[[\n    'PlayerKey',\n    'RosterPosition',\n    'PlayerGamePlay',\n    'Position',\n    'PositionGroup'\n]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_player_features(player_df, figsize=(25, 20), add_labels=False):\n    plt.style.use('ggplot')\n    fig = plt.figure(figsize=figsize)\n    \n    grid = plt.GridSpec(3, 4, hspace=0.2, wspace=0.2)\n    \n    plays_ax = fig.add_subplot(grid[0, :2])\n    roster_position_ax = fig.add_subplot(grid[0, 2:])\n    max_rolling_plays_ax = fig.add_subplot(grid[1, :2])\n    position_group_ax = fig.add_subplot(grid[1, 2:])\n    position_ax = fig.add_subplot(grid[2, :])\n    \n    plays_ax.hist(\n        player_df.groupby(by=['PlayerKey']).count()['RosterPosition'].values,\n        bins=20,\n        color='#00c2c7'\n    )\n    plays_ax.set_title('Number of plays per player')\n    \n    roster_position_ax.bar(\n        player_df.RosterPosition.value_counts().keys().values,\n        player_df.RosterPosition.value_counts().values\n    )\n    roster_position_ax.set_xticklabels(\n        player_df.RosterPosition.value_counts().keys().values,\n        rotation=20\n    )\n    roster_position_ax.set_title('Roster Position')\n    if add_labels:\n        add_value_labels(roster_position_ax, spacing=5)\n    \n    max_rolling_plays_ax.hist(\n        player_df.groupby(by=['PlayerKey']).PlayerGamePlay.max().values,\n        bins=20,\n        color='#00c2c7'\n    )\n    max_rolling_plays_ax.set_title('Maximum number of rolling plays per player')\n    \n    position_group_ax.bar(\n        player_df.PositionGroup.value_counts().keys().values,\n        player_df.PositionGroup.value_counts().values\n    )\n    position_group_ax.set_title('Position Group')\n    if add_labels:\n        add_value_labels(position_group_ax, spacing=5)\n    \n    position_ax.bar(\n        player_df.Position.value_counts().keys().values,\n        player_df.Position.value_counts().values,\n        color='#ff9e15'\n    )\n    position_ax.set_title('Position')\n    if add_labels:\n        add_value_labels(position_ax, spacing=5)\n    \n    plt.suptitle('Player-Level Exploration', fontsize=16)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_player_features(player_data_df, add_labels=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Nothing could be infered from the above graphs"},{"metadata":{},"cell_type":"markdown","source":"#### Play-level exploration"},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_play(play_df, figsize=(15, 5), add_labels=True):\n    plt.style.use('ggplot')\n    fig, ax = plt.subplots(1, 1, figsize=figsize)\n    \n    plt.bar(\n        play_df.PlayType.value_counts().keys().values,\n        play_df.PlayType.value_counts().values\n    )\n    plt.xticks(\n        range(len(play_df.PlayType.value_counts().keys().values)),\n        play_df.PlayType.value_counts().keys().values,\n        rotation=20\n    )\n    if add_labels:\n        add_value_labels(ax, spacing=5)\n    plt.title('Play-Level Exploration: PlayType', fontsize=16)\n    \n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_play(play_df_cleaned)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The huge difference might suggest that the play-level might affect injuries."},{"metadata":{},"cell_type":"markdown","source":"## Player Dataset Visualization and Exploration"},{"metadata":{},"cell_type":"markdown","source":"#### Visualize players positions and paths:"},{"metadata":{},"cell_type":"markdown","source":"Visaulize one line of the dataset:"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_position(player_df, play_key, time):\n    row = player_df[(player_df['PlayKey'] == play_key) & (player_df['time'] == time)]\n    event = row['event'].values[0]\n    x = row['x'].values[0]\n    y = row['y'].values[0]\n    direction = row['dir'].values[0]\n    distance = row['dis'].values[0]\n    orientation = row['o'].values[0]\n    speed = row['s'].values[0]\n    \n    return event, x, y, direction, distance, orientation, speed\n\n\ndef visualize_player_position(player_df, play_key, time, figsize=(24, 10)):\n    event, x, y, direction, distance, orientation, speed = get_position(player_df, play_key, time)\n    \n    fig = plt.figure(figsize=figsize)\n    \n    dx = 5\n    dy = dx * np.tan(np.radians(90 + orientation))\n    plt.arrow(x*10, y*10, dx, dy, color='#767676', width=5)\n    plt.plot(x*10, y*10, color='#767676', label='orientation')\n    \n    dx = speed * 20\n    dy = dx * np.tan(np.radians(90 + direction))\n    plt.arrow(x * 10, y * 10, dx, dy, color='#004c97', width=5)\n    plt.plot(x * 10, y * 10, color='#004c97', label='speed')\n    \n    plt.scatter(x * 10, y * 10, s=200, color='#e01e5a', marker='x')\n    plt.annotate(\n        '({x:.1f},{y:.1f})'.format(x=x, y=y),\n        (x*10, y*10),\n        xytext=(x*10, y*10-30),\n        color='#e01e5a'\n    )\n    \n    plt.xticks(range(0, 1200, 100), range(0, 120, 10))\n    plt.yticks(range(0, 533, 100), range(0, 53, 10))\n    \n    plt.title('{play}:{time} {event}'.format(\n        play=play_key,\n        time=time,\n        event=event\n    ))\n    \n    plt.legend()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_player_position(player_df, '26624-1-1', 0.0, figsize=(10, 5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_player_position(player_df, '47888-13-55', 35.9, figsize=(10, 5))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Plot the entire path of the player during the play."},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_player_track(player_df, play_key, figsize=(24, 10), max_positions=5):\n    fig = plt.figure(figsize=figsize)\n    \n    timestamps = player_df[player_df['PlayKey'] == play_key].time.unique()\n    positions_x, positions_y = [], []\n    for i in range(0, len(timestamps), len(timestamps) // max_positions):\n        time = timestamps[i]\n        event, x, y, direction, distance, orientation, speed = get_position(player_df, play_key, time)\n        positions_x.append(x * 10)\n        positions_y.append(y * 10)\n        \n        if (len(timestamps) - i < len(timestamps) // max_positions):\n            dx = 4\n            dy = dx * np.tan(np.radians(90 + orientation))\n            plt.arrow(\n                x * 10,\n                y * 10,\n                dx, dy,\n                color='#767676',\n                width=5\n            )\n            plt.plot(\n                x * 10,\n                y * 10,\n                color='#767676',\n                label='orientation'\n            )\n            \n            dx = speed * 20\n            dy = dx * np.tan(np.radians(90 + direction))\n            plt.arrow(\n                x * 10,\n                y * 10,\n                dx, dy,\n                color='#004c97',\n                width=5\n            )\n            plt.plot(\n                x * 10,\n                y * 10,\n                color='#004c97',\n                label='speed'\n            )\n            \n            plt.scatter(\n                x * 10,\n                y * 10,\n                s=200,\n                color='#e01e5a',\n                marker='x'\n            )\n            plt.annotate(\n                '({x:.1f},{y:.1f})'.format(x=x, y=y),\n                (x*10, y*10),\n                xytext=(x*10, y*10-30),\n                color='#e01e5a'\n            )\n    \n    plt.scatter(\n        positions_x, \n        positions_y, \n        s=50, \n        color='#e01e5a', \n        marker='o'\n    )\n    plt.plot(\n        positions_x,\n        positions_y,\n        color='#e01e5a',\n        label='player_path',\n        linestyle='--'\n    )\n    \n    plt.xticks(range(0, 1200, 100), range(0, 120, 10))\n    plt.yticks(range(0, 533, 100), range(0, 53, 10))\n    \n    plt.title('{play}'.format(play=play_key))\n    \n    plt.legend()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_player_track(\n    player_df, \n    '26624-1-1', \n    max_positions=5, \n    figsize=(10, 5)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_player_track(\n    player_df,\n    '47888-13-55',\n    max_positions=5,\n    figsize=(10, 5)\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Create a heatmap of the field"},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_field_heatmap(player_df, xbins=13, ybins=6, annotate=False):\n    x = np.linspace(0, 120, xbins)\n    y = np.linspace(0, 53, ybins)\n    \n    hmap = np.zeros((xbins, ybins))\n    \n    for i in range(xbins-1):\n        for j in range(ybins-1):\n            hmap[i, j] = len(player_df[(player_df.x >= x[i]) & (player_df.x <= x[i+1]) & (player_df.y >= y[j]) & (player_df.y <= y[j+1])])\n            \n    fig = plt.figure(figsize=(10, 5))\n    ax = sns.heatmap(np.transpose(hmap), annot=annotate, fmt='.0f')\n    plt.title('Field Heatmap \\n the most visited areas of the field are highlighted')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"visualize_field_heatmap(player_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Some areas of the field are more busy than the others."},{"metadata":{},"cell_type":"markdown","source":"#### Injuries Dataset EDA"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Total number of injury records: {}'.format(len(injury_df)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number of unique players injured: {}'.format(len(injury_df.PlayerKey.unique())))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This could mean that there are players injured twice."},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number of missing PlayKey values: {}'.format(len(injury_df) - injury_df.PlayKey.count()))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"That is a big number when compared to the available number of players."},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_injury(injury_df, figsize=(15, 5), add_labels=True):\n    fig, axs = plt.subplots(1, 3, figsize=figsize)\n    \n    axs[0].bar(\n        injury_df.BodyPart.value_counts().keys().values,\n        injury_df.BodyPart.value_counts().values,\n        color='#00c2c7'\n    )\n    axs[0].set_title('Body Part')\n    add_value_labels(axs[0], spacing=5)\n    \n    axs[1].bar(\n        injury_df.Surface.value_counts().keys().values,\n        injury_df.Surface.value_counts().values,\n        color='#ff9e15'\n    )\n    axs[1].set_title('Surface')\n    if add_labels:\n        add_value_labels(axs[1], spacing=5)\n    \n    M1 = injury_df.DM_M1.sum()\n    M7 = injury_df.DM_M7.sum()\n    M28 = injury_df.DM_M28.sum()\n    M42 = injury_df.DM_M42.sum()\n    \n    axs[2].bar(['1-7', '7-28', '28-42', '>=42'], [M1, M7, M28, M42])\n    axs[2].set_title('Missed Days')\n    if add_labels:\n        add_value_labels(axs[2], spacing=5)\n    \n    plt.suptitle('Injury', fontsize=16)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"injury_df_cleaned = injury_df.copy()\ninjury_df_cleaned.DM_M1 = injury_df_cleaned.DM_M1 - injury_df_cleaned.DM_M7\ninjury_df_cleaned.DM_M7 = injury_df_cleaned.DM_M7 - injury_df_cleaned.DM_M28\ninjury_df_cleaned.DM_M28 = injury_df_cleaned.DM_M28 - injury_df_cleaned.DM_M42\nvisualize_injury(injury_df_cleaned)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* There is very less number of injury records.\n* Knees and ankles are the most frequently injured body parts.\n* Almost equal number is observed for synthetic and natural surface."},{"metadata":{},"cell_type":"markdown","source":"## Injury Analysis"},{"metadata":{},"cell_type":"markdown","source":"#### injuries vs game-level factors"},{"metadata":{"trusted":true},"cell_type":"code","source":"# join cleaned games df with injury df\ngame_injury_df = injury_df.set_index('GameID').join(\n                    game_df_cleaned.set_index('GameID'),\n                    how='outer'\n                )\n\n# fill null values for the injury columns with zeros\ngame_injury_df['DM_M1'] = game_injury_df['DM_M1'].fillna(0).astype(int)\ngame_injury_df['DM_M7'] = game_injury_df['DM_M7'].fillna(0).astype(int)\ngame_injury_df['DM_M28'] = game_injury_df['DM_M28'].fillna(0).astype(int)\ngame_injury_df['DM_M42'] = game_injury_df['DM_M42'].fillna(0).astype(int)\n\ngame_injury_df.DM_M1 = game_injury_df.DM_M1 - game_injury_df.DM_M7\ngame_injury_df.DM_M7 = game_injury_df.DM_M7 - game_injury_df.DM_M28\ngame_injury_df.DM_M28 = game_injury_df.DM_M28 - game_injury_df.DM_M42\n\n# introduce flag indication an injury\ngame_injury_df['Injury'] = game_injury_df['DM_M1'] + game_injury_df['DM_M7'] + game_injury_df['DM_M28'] + game_injury_df['DM_M42']\n# drop duplicated surface column\ngame_injury_df = game_injury_df.drop(columns=['Surface'])\n# drop play-level features\ngame_injury_df = game_injury_df.drop(columns=['PlayerKey', 'PlayKey'])\n# create dummy variables\ngame_injury_df_dummies = pd.get_dummies(\n                            game_injury_df,\n                            dummy_na=True,\n                            drop_first=True\n                        ).drop(columns=['FieldType_nan'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"game_injury_corr_df = game_injury_df_dummies[[\n                            'Temperature', 'StadiumType_Indoor', \n                            'StadiumType_Open', 'StadiumType_Outdoor', \n                            'StadiumType_Retractable Roof', 'FieldType_Synthetic', \n                            'Weather_Cloudy', 'Weather_Rain',\n                            'Weather_Snow', 'Injury'\n                        ]].corr()\n\nfig = plt.figure(figsize=(10, 7))\nsns.heatmap(\n    game_injury_corr_df, annot=True, \n    cmap=sns.diverging_palette(220, 20, as_cmap=True)\n)\nplt.title('Correlation Heatmap')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**there is no much correlation between the features and the injury**"},{"metadata":{},"cell_type":"markdown","source":"Let's look at [Cramer's V](https://en.wikipedia.org/wiki/Cram%C3%A9r%27s_V):"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Source:\n# https://stackoverflow.com/questions/46498455/categorical-features-correlation/46498792#46498792\ndef cramers_v(confusion_matrix):\n    chi2 = ss.chi2_contingency(confusion_matrix)[0]\n    n = confusion_matrix.sum()\n    phi2 = chi2 / n\n    r, k = confusion_matrix.shape\n    phi2corr = max(0, phi2 - ((k-1)*(r-1))/(n-1))\n    rcorr = r - ((r-1)**2)/(n-1)\n    kcorr = k - ((k-1)**2)/(n-1)\n    return np.sqrt(phi2corr / min((kcorr-1), (rcorr-1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_cramers_v_heatmap(df, cols):\n    corrM = np.zeros((len(cols), len(cols)))\n    \n    for col1, col2 in itertools.combinations(cols, 2):\n        idx1, idx2 = cols.index(col1), cols.index(col2)\n        corrM[idx1, idx2] = cramers_v(pd.crosstab(df[col1], df[col2]).values)\n        corrM[idx2, idx1] = corrM[idx1, idx2]\n    \n    corr = pd.DataFrame(corrM, index=cols, columns=cols)\n    fig, ax = plt.subplots(figsize=(7, 6))\n    ax = sns.heatmap(corr, annot=True, cmap=sns.diverging_palette(220, 20, as_cmap=True), ax=ax)\n    ax.set_title(\"Cramer V Correlation between Variables\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cols = ['StadiumType', 'FieldType', 'Weather', 'Temperature', 'Injury']\nplot_cramers_v_heatmap(game_injury_df, cols)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* there are correlations between Weather and Temperature\n* there is some correlation between Stadium Type and Temperature\n* but, there is no correlation with the injury"},{"metadata":{},"cell_type":"markdown","source":"[Theil's U](https://docs.oracle.com/cd/E40248_01/epm.1112/cb_statistical/frameset.htm?ch07s02s03s04.html) to find some more insights"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Source:\n# https://github.com/shakedzy/dython/blob/master/dython/nominal.py\nfrom collections import Counter\nimport math\n\n\ndef conditional_entropy(x, y, nan_strategy='replace', nan_replace_value=0):\n    y_counter = Counter(y)\n    xy_counter = Counter(list(zip(x, y)))\n    total_occurrences = sum(y_counter.values())\n    entropy = 0.0\n    for xy in xy_counter.keys():\n        p_xy = xy_counter[xy] / total_occurrences\n        p_y = y_counter[xy[1]] / total_occurrences\n        entropy += p_xy * math.log(p_y / p_xy)\n    return entropy\n\n\ndef theils_u(x, y):\n    s_xy = conditional_entropy(x, y)\n    x_counter = Counter(x)\n    total_occurrences = sum(x_counter.values())\n    p_x = list(map(lambda n: n/total_occurrences, x_counter.values()))\n    s_x = ss.entropy(p_x)\n    if s_x == 0:\n        return 1\n    return (s_x - s_xy) / s_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_theils_u_heatmap(df, cols):\n    corrM = np.zeros((len(cols), len(cols)))\n    \n    for col1, col2 in itertools.combinations(cols, 2):\n        idx1, idx2 = cols.index(col1), cols.index(col2)\n        corrM[idx1, idx2] = theils_u(df[col1].values, df[col2].values)\n        corrM[idx2, idx1] = corrM[idx1, idx2]\n    \n    corr = pd.DataFrame(corrM, index=cols, columns=cols)\n    fig, ax = plt.subplots(figsize=(7, 6))\n    ax = sns.heatmap(\n        corr, annot=True, \n        cmap=sns.diverging_palette(\n            220, 20, \n            as_cmap=True\n        ), ax=ax)\n    ax.set_title(\"Theil's U Correlation between Variables\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cols = [\"StadiumType\", \"FieldType\", \"Weather\", \"Temperature\", \"Injury\"]\nplot_theils_u_heatmap(game_injury_df, cols)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"temperature vs injury"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(10, 5))\nplt.boxplot([\n    game_injury_df[game_injury_df.Injury == 0].Temperature.values,\n    game_injury_df[game_injury_df.Injury == 1].Temperature.values\n], vert=False)\nplt.title('Temperature Distribution')\nplt.yticks([1, 2], ['No Injury', 'Injury'])\nplt.xlim(0, 100)\nplt.xlabel('Temperature')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Nothing significant changes can be observed because of less number of injury data."},{"metadata":{},"cell_type":"markdown","source":"Play vs Injury"},{"metadata":{"trusted":true},"cell_type":"code","source":"# join cleaned games dataset and injury dataset\nplay_injury_df = injury_df.dropna(subset=['PlayKey']).set_index('PlayKey').join(\n    play_df_cleaned.set_index('PlayKey'),\n    how='outer', lsuffix='_left', rsuffix='_right'\n)\n\n# fill null values\nplay_injury_df['DM_M1'] = play_injury_df['DM_M1'].fillna(0).astype(int)\nplay_injury_df['DM_M7'] = play_injury_df['DM_M7'].fillna(0).astype(int)\nplay_injury_df['DM_M28'] = play_injury_df['DM_M28'].fillna(0).astype(int)\nplay_injury_df['DM_M42'] = play_injury_df['DM_M42'].fillna(0).astype(int)\n\n# compute the columns\nplay_injury_df.DM_M1 = play_injury_df.DM_M1 - play_injury_df.DM_M7\nplay_injury_df.DM_M7 = play_injury_df.DM_M7 - play_injury_df.DM_M28\nplay_injury_df.DM_M28 = play_injury_df.DM_M28 - play_injury_df.DM_M42\n# injury column\nplay_injury_df['Injury'] = play_injury_df['DM_M1'] + play_injury_df['DM_M7'] + play_injury_df['DM_M28'] + play_injury_df['DM_M42']\n\nplay_injury_df = play_injury_df.drop(columns=['Surface'])\nplay_injury_df_dummies = pd.get_dummies(play_injury_df, columns=['PlayType', 'PositionGroup'], dummy_na=True, drop_first=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"corr_df = play_injury_df_dummies[[\n    'PlayType_Pass', 'PlayType_Kickoff', 'PlayType_Punt', 'PlayType_Rush',\n    'PositionGroup_QB', 'PositionGroup_DL', 'PositionGroup_LB', \n    'PositionGroup_OL', 'PositionGroup_RB', 'PositionGroup_SPEC', \n    'PositionGroup_TE', 'PositionGroup_WR', 'Injury'\n]].corr()\n\nfig = plt.figure(figsize=(15, 7))\nsns.heatmap(corr_df, annot=True, cmap=sns.diverging_palette(220, 20, as_cmap=True))\nplt.title('Correlation Heatmap')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Cramer's V:"},{"metadata":{"trusted":true},"cell_type":"code","source":"cols = ['RosterPosition', 'StadiumType', 'FieldType', \n        'Temperature', 'Weather', 'PlayType', 'PlayerGamePlay', \n        'Position', 'PositionGroup', 'Injury']\nplot_cramers_v_heatmap(play_injury_df, cols)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Theil's U:"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_theils_u_heatmap(play_injury_df, cols)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Visualize the heatmap of the field with the plays with the injury:"},{"metadata":{"trusted":true},"cell_type":"code","source":"play_injuries = play_injury_df.reset_index().dropna()[['PlayKey']]\nplayer_injuries = player_df.merge(play_injuries, on='PlayKey', how='inner')\nvisualize_field_heatmap(player_injuries, annotate=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"the heatmap of injuries is somewhat different from the general heatmap of the field! This may mean that there are more dangerous areas on the field, where players are more likely to get injured!\n\nWe can use this insight for the feature engineering for the injury prediction model!"},{"metadata":{},"cell_type":"markdown","source":"#### game timeline and injury analysis"},{"metadata":{"trusted":true},"cell_type":"code","source":"def player_games_timeline(player_key, play_df, injury_df):\n    player_games = play_df[play_df.PlayerKey == player_key][[\n        'GameID', 'PlayKey', 'PlayerDay', 'PlayerGame'\n    ]]\n    \n    plt.figure(figsize=(20, 5))\n    plt.title('Player Games Timeline \\n PlayerKey: {:d}'.format(player_key))\n    plt.plot(\n        player_games.PlayerDay.unique(),\n        np.zeros(len(player_games.PlayerDay.unique())),\n        color='#00c2c7'\n    )\n    plt.scatter(\n        player_games.PlayerDay.unique(),\n        np.zeros(len(player_games.PlayerDay.unique())),\n        s=100, color='#00c2c7', label='games'\n    )\n    \n    injured_players = injury_df.PlayerKey.unique()\n    if player_key in injured_players:\n        injury_games = injury_df[injury_df.PlayerKey == player_key].GameID.values\n        injury_days = player_games[player_games.GameID.isin(injury_games)].PlayerDay.unique()\n        \n        plt.scatter(\n            injury_days, \n            np.zeros(len(injury_days)), \n            s=100, color='#e01e5a', label='injury'\n        )\n    \n    plt.legend()\n    plt.xlabel('days')\n    plt.yticks([])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"player_games_timeline(26624, play_df, injury_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"player_games_timeline(33337, play_df, injury_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"player_games_timeline(43540, play_df, injury_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"player_games_timeline(39873, play_df, injury_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Looks like most of the time, the injury happens at the beginning of the season."},{"metadata":{},"cell_type":"markdown","source":"Is the speed of the player, his position on the field and the number of the game (```PlayerGame```) correlated with the injury?"},{"metadata":{"trusted":true},"cell_type":"code","source":"play_injury = play_injury_df[['PlayerGame', 'PlayerGamePlay', 'Injury']]\ncorrs = play_injury.corr()\n\nfig = plt.figure(figsize=(7, 5))\nsns.heatmap(\n    corrs, annot=True,\n    cmap=sns.diverging_palette(220, 20, as_cmap=True)\n)\nplt.title('Correlation Heatmap')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There is no correlation between the injury and the number of the games/plays per game played."},{"metadata":{},"cell_type":"markdown","source":"## Data Engineering\n\nExtract features and try to build a model to predic the injury."},{"metadata":{"trusted":true},"cell_type":"code","source":"features_df = play_injury_df.copy().reset_index()\nfeatures_df = features_df.drop(\n    columns=[\n        'PlayerKey_left', 'GameID_left', 'BodyPart', \n        'PlayKey', 'PlayerKey_right', 'GameID_right',\n        'DM_M1', 'DM_M7', 'DM_M28', 'DM_M42'\n    ]\n)\nfeatures_df = pd.get_dummies(\n    features_df,\n    dummy_na=False,\n    drop_first=True\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Machine Learning"},{"metadata":{},"cell_type":"markdown","source":"#### Split into features and targets:"},{"metadata":{"trusted":true},"cell_type":"code","source":"y = features_df['Injury']\nX = features_df.drop(columns=['Injury'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### split into train and test"},{"metadata":{"trusted":true},"cell_type":"code","source":"skf = StratifiedKFold(n_splits=2)\n\nfor train_index, test_index in skf.split(X, y):\n    X_train, X_test = X.values[train_index, :], X.values[test_index, :]\n    y_train, y_test = y[train_index], y[test_index]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Resampling the dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"res = RandomOverSampler(random_state=0)\nX_resampled, y_resampled = res.fit_resample(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Train the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = xgb.XGBClassifier(max_depth=3,\n                      learning_rate=0.1,\n                      n_estimators=100,\n                      objective='binary:logistic',\n                      booster='gbtree',\n                      tree_method='auto',\n                      n_jobs=50,\n                      gamma=0,\n                      min_child_weight=1,\n                      max_delta_step=0,\n                      subsample=1,\n                      colsample_bytree=1,\n                      colsample_bylevel=1,\n                      colsample_bynode=1,\n                      reg_alpha=0,\n                      reg_lambda=1,\n                      scale_pos_weight=1,\n                      base_score=0.5,\n                      random_state=42)\nmodel.fit(X_resampled, y_resampled)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Evaluate the model:"},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = model.predict(X_test)\n\naccuracy = accuracy_score(y_test, y_pred)\nconf_matrix = confusion_matrix(y_test, y_pred)\ncohen_kappa = cohen_kappa_score(y_test, y_pred)\n\nprint('Accuracy: {}'.format(accuracy))\nprint('Cohen kappa: {}'.format(cohen_kappa))\nprint('Confusion Matrix: \\n {}'.format(conf_matrix))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Model's performance is quite poor, but we can still look at the feature importances"},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_importances = model.feature_importances_\nfeature_importances = pd.DataFrame(\n    feature_importances, \n    index=X.columns\n).reset_index().rename(columns={\n    'index': 'feature',\n    0: 'importance'\n}).sort_values(by=['importance'], ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20, 7))\nplt.bar(range(len(feature_importances)), feature_importances.importance.values)\nplt.xticks(range(len(feature_importances)), feature_importances.feature.values, rotation=90)\n\nplt.title('Feature importances')\nplt.xlabel('features')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* weather and stadium type are possibly important factors impacting injury\n* field type feature is not among the important features in our injury predicting model\n* player day and player game are actually important"},{"metadata":{},"cell_type":"markdown","source":"## Motion Data Analysis"},{"metadata":{},"cell_type":"markdown","source":"Different features for each play can be extracted:\n* the average speed per play\n* the sharpest turn per play\n* the average angle between the direction and the orientation\n* the maximum angle between the direction and the orientation"},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_motion_data_df(injury_df, play_df, player_df):\n    player_df['angle'] = player_df['o'] - player_df['dir']\n    player_df['speed'] = player_df['s']\n    \n    grouped_max = player_df[['PlayKey', 'time', 'dir', 'dis', 'o', 's', 'speed', 'angle']].groupby(by=['PlayKey']).max()\n    grouped_average = player_df[['PlayKey', 'time', 'dir', 'dis', 'o', 's', 'speed', 'angle']].groupby(by=['PlayKey']).mean()\n    \n    play_df = play_df.merge(grouped_max.reset_index(), on=['PlayKey'])\n    play_df = play_df.merge(grouped_average.reset_index(), on=['PlayKey'], suffixes=('_max', 'avg'))\n    \n    injury_df = injury_df.drop(columns=['PlayerKey', 'GameID', 'BodyPart', 'Surface']).merge(play_df, on=['PlayKey'], how='outer').fillna(0)\n    \n    return injury_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"motion_df = create_motion_data_df(injury_df_cleaned, play_df_cleaned, player_df)\nmotion_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"player_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## References and Credits\n* [NFL Injury Analysis](https://www.kaggle.com/aleksandradeis/nfl-injury-analysis) notebook by [Aleksandra Deis](https://www.kaggle.com/aleksandradeis)"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}