{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\npd.plotting.register_matplotlib_converters()\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-10-29T08:15:42.501493Z","iopub.execute_input":"2021-10-29T08:15:42.501792Z","iopub.status.idle":"2021-10-29T08:15:42.515892Z","shell.execute_reply.started":"2021-10-29T08:15:42.50176Z","shell.execute_reply":"2021-10-29T08:15:42.515036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **NFL Big Data Bowl 2022**\n\n#### Punt (пант) является одним из самых важных действий специальной командой. Понимание того, что приводит к получению очков после исполнения панта является одной из ключевых задач для тренерского состава. Данный блокнот содержит попытку создания модели предсказывающую вероятность того каким будет результат исполнения панта. ","metadata":{}},{"cell_type":"markdown","source":"<center>\n<img src=\"https://i.gifer.com/8Zr9.gif\" alt=\"drawing\"/>\n<img src=\"https://static.www.nfl.com/image/upload/v1554321393/league/nvfr7ogywskqrfaiu38m.svg\" alt=\"drawing\" width=\"320\"/>\n</center>","metadata":{}},{"cell_type":"markdown","source":"## Обработка исходных данных. Выбор цели и фичей для модели","metadata":{}},{"cell_type":"code","source":"#Загрузка данных\n#-------------------------------\ngames_path = '../input/nfl-big-data-bowl-2022/games.csv'\nplays_path = '../input/nfl-big-data-bowl-2022/plays.csv'\nplayers_path = '../input/nfl-big-data-bowl-2022/players.csv'\nPFFScoutingData_path = '../input/nfl-big-data-bowl-2022/PFFScoutingData.csv'\ntracking2018_path = '../input/nfl-big-data-bowl-2022/tracking2018.csv'\ntracking2019_path = '../input/nfl-big-data-bowl-2022/tracking2019.csv'\ntracking2020_path = '../input/nfl-big-data-bowl-2022/tracking2020.csv'\n\ngames_data = pd.read_csv(games_path)\nplays_data = pd.read_csv(plays_path)\nplayers_data = pd.read_csv(players_path)\nPFFScoutingData_data = pd.read_csv(PFFScoutingData_path)\n#tracking2018_data = pd.read_csv(tracking2018_path)\n#tracking2019_data = pd.read_csv(tracking2019_path)\n#tracking2020_data = pd.read_csv(tracking2020_path)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.517954Z","iopub.execute_input":"2021-10-29T08:15:42.518789Z","iopub.status.idle":"2021-10-29T08:15:42.671218Z","shell.execute_reply.started":"2021-10-29T08:15:42.518738Z","shell.execute_reply":"2021-10-29T08:15:42.670588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Посмотрим что из себя представлют данные предоставленные NFL\n#В первую очередь рассмотрим данные связанные с мячом, содержающиеся в plays_path и PFFScoutingData_path\n#-------------------------------\ndef describe_data(data, n = 10, d = False):\n    print(data.shape)\n    print(data.columns)\n    print()\n    if d == True:\n        print(data.describe())\n    print()\n    return data.head(n)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.672899Z","iopub.execute_input":"2021-10-29T08:15:42.673368Z","iopub.status.idle":"2021-10-29T08:15:42.679223Z","shell.execute_reply.started":"2021-10-29T08:15:42.673329Z","shell.execute_reply":"2021-10-29T08:15:42.678409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe_data(plays_data)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-10-29T08:15:42.680906Z","iopub.execute_input":"2021-10-29T08:15:42.681175Z","iopub.status.idle":"2021-10-29T08:15:42.718754Z","shell.execute_reply.started":"2021-10-29T08:15:42.681147Z","shell.execute_reply":"2021-10-29T08:15:42.71819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe_data(PFFScoutingData_data)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-10-29T08:15:42.720092Z","iopub.execute_input":"2021-10-29T08:15:42.720565Z","iopub.status.idle":"2021-10-29T08:15:42.746631Z","shell.execute_reply.started":"2021-10-29T08:15:42.720534Z","shell.execute_reply":"2021-10-29T08:15:42.74587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Будем рассматривать только те фичи, что имеют отношение к мячу (его положение, расстояние на которое его пнули и тп.). Все остальные, в том числе из других наборов данных не рассматриваем.  \n> **specialTeamsPlayType**: Formation of play: Extra Point, Field Goal, Kickoff or Punt (text)  \n> **specialTeamsResult**: Special Teams outcome of play dependent on play type: Blocked Kick Attempt, Blocked Punt, Downed, Fair Catch, Kick Attempt Good, Kick Attempt No Good, Kickoff Team Recovery, Muffed, Non-Special Teams Result, Out of Bounds, Return or Touchback (text)  \n> **yardlineNumber**: Yard line at line-of-scrimmage (numeric)  \n> **kickLength**: Kick length in air of kickoff, field goal or punt (numeric)  \n> **kickReturnYardage**: Yards gained by return team if there was a return on a kickoff or punt (numeric)  \n> **playResult**: Net yards gained by the kicking team, including penalty yardage (numeric)  \n> **absoluteYardlineNumber**: Location of ball downfield in tracking data coordinates (numeric)  \n> **snapDetail**: On Punts, whether the snap was on target and if not, provides detail (H: High, L: Low, <: Left, >: Right, OK: Accurate Snap, text)  \n> **kickType**: Kickoff or Punt Type (text). Possible values for punt plays:   \n> > N: Normal - standard punt style  \n> > R: Rugby style punt  \n> > A: Nose down or Aussie-style punts  \n\n> **kickContactType**: Detail on how a punt was fielded, or what happened when it wasn't fielded (text). Possible values:  \n> > BB: Bounced Backwards  \n> > BC: Bobbled Catch from Air  \n> > BF: Bounced Forwards  \n> > BOG: Bobbled on Ground  \n> > CC: Clean Catch from Air  \n> > CFFG: Clean Field From Ground  \n> > DEZ: Direct to Endzone  \n> > ICC: Incidental Coverage Team Contact  \n> > KTB: Kick Team Knocked Back  \n> > KTC: Kick Team Catch  \n> > KTF: Kick Team Knocked Forward  \n> > MBC: Muffed by Contact with Non-Designated Returner  \n> > MBDR: Muffed by Designated Returner  \n> > OOB: Directly Out Of Bounds  \n\n> **operationTime**: Timing from snap to kick on punt plays in seconds: (numeric)  \n> **hangTime**: Hangtime of player's punt or kickoff attempt in seconds. Timing is taken from impact with foot to impact with the ground or a player. (numeric)\n\n","metadata":{}},{"cell_type":"code","source":"#выбранные фичи из Plays\ncol_plays_use_cat = ['specialTeamsPlayType', 'specialTeamsResult']\ncol_plays_use_num = ['yardlineNumber', 'kickLength', 'kickReturnYardage', 'playResult', 'absoluteYardlineNumber']\n\n#выбранные фичи из PFFScoutingData\ncol_PFF_use_cat = ['snapDetail', 'kickType', 'kickContactType']\ncol_PFF_use_num = ['operationTime', 'hangTime']","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.747876Z","iopub.execute_input":"2021-10-29T08:15:42.748101Z","iopub.status.idle":"2021-10-29T08:15:42.752966Z","shell.execute_reply.started":"2021-10-29T08:15:42.74807Z","shell.execute_reply":"2021-10-29T08:15:42.752189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Объединение DF для работы\n#-------------------------------\ndef creater_df():\n    df1_temp = plays_data[['gameId', 'playId',] + col_plays_use_cat + col_plays_use_num].copy().set_index(['gameId', 'playId'])\n    df2_temp = PFFScoutingData_data[['gameId', 'playId',] + col_PFF_use_cat + col_PFF_use_num].copy().set_index(['gameId', 'playId'])\n    df_temp = df1_temp.join(df2_temp, lsuffix='_CAN', rsuffix='_UK')\n    return df_temp","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.754274Z","iopub.execute_input":"2021-10-29T08:15:42.7547Z","iopub.status.idle":"2021-10-29T08:15:42.763343Z","shell.execute_reply.started":"2021-10-29T08:15:42.754664Z","shell.execute_reply":"2021-10-29T08:15:42.762603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Определим цель модели. Посмотрим на результаты действий специальных команд. Будущая модель задумывается для предсказания результата розыгрыша панта. Для него есть восемь исходов, каждый из них случался совершенно разное кол-во раз за все игры представленные в данных. Можно предположить, что модель сможет их различать, а значит выбор целью модели specialTeamsResult оправдан. ","metadata":{}},{"cell_type":"code","source":"#Просмотр статистики по действиям специальных команд\n#-------------------------------\ndf = creater_df()\ndf_temp = pd.DataFrame(df.groupby(['specialTeamsPlayType', 'specialTeamsResult']).specialTeamsResult.count())\ndf_temp\n","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.764596Z","iopub.execute_input":"2021-10-29T08:15:42.764926Z","iopub.status.idle":"2021-10-29T08:15:42.796499Z","shell.execute_reply.started":"2021-10-29T08:15:42.764889Z","shell.execute_reply":"2021-10-29T08:15:42.795808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Создание DFs для работы. DF будет содержать данные только для Punt\n#-------------------------------\n\ndf = creater_df()\ndf_punt = df.loc[df['specialTeamsPlayType'].isin(['Punt'])]\n\n#Проверка уникальности данных в столбцах\ndef unique_incol(data):\n    for col in data.columns:\n        print(data[col].name)\n        print(data[col].count())\n        print(data[col].isnull().sum(axis = 0))\n        print(data[col].unique(), \"\\n\")\n\n#unique_incol(df_punt)\n#print(\"---------------------------------------\")\n\n#Выбросим значения по строкам, которые невозможно заполнить (категориальные)\ndf_punt = df_punt.dropna(subset=['kickType', 'kickContactType', 'snapDetail'])\n#kickReturnYardage имет 60% значений NaN, выбросим данный столбец на текущем этапе\ndf_punt.drop(columns = ['kickReturnYardage'], inplace=True)\n#Заполним оставшиеся значения NaN средними значениями по строкам\ndf_punt['operationTime'] = round(df_punt['operationTime'].fillna(df_punt['operationTime'].mean()), 2)\ndf_punt['hangTime'] = round(df_punt['hangTime'].fillna(df_punt['hangTime'].mean()), 2)\n\nunique_incol(df_punt)\ndf_punt.index = df_punt.index.droplevel(0)\ndf_punt","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.798833Z","iopub.execute_input":"2021-10-29T08:15:42.799274Z","iopub.status.idle":"2021-10-29T08:15:42.868227Z","shell.execute_reply.started":"2021-10-29T08:15:42.799237Z","shell.execute_reply":"2021-10-29T08:15:42.867412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Визуализация данных  \n","metadata":{}},{"cell_type":"markdown","source":"#### Посмотрим на числовые фичи для каждого исхода панта. Для этого построим скрипичные графики для числовых фичей, а категориальные сгрупируем в компактные таблица по количеству характеристик для каждого исхода.\n#### Данные действия позволят оценить насколько каждый исход для разных фичей отличается и есть ли смысл использовать эти фичи. ","metadata":{}},{"cell_type":"code","source":"#Визуализация соотношений результатов специальных команд и числовых характеристик\n#-------------------------------\nfig = plt.figure(figsize=(18, 15))\ngs = fig.add_gridspec(3, 2)\n\nax = fig.add_subplot(gs[0, 0])\nsns.violinplot(data = df_punt, x='specialTeamsResult', y='yardlineNumber')\n\nax = fig.add_subplot(gs[0, 1])\nsns.violinplot(data = df_punt, x='specialTeamsResult', y='kickLength')\n\nax = fig.add_subplot(gs[1, 0])\nsns.violinplot(data = df_punt, x='specialTeamsResult', y='playResult')\n\nax = fig.add_subplot(gs[1, 1])\nsns.violinplot(data = df_punt, x='specialTeamsResult', y='absoluteYardlineNumber')\n\nax = fig.add_subplot(gs[2, 0])\nsns.violinplot(data = df_punt, x='specialTeamsResult', y='operationTime')\n\nax = fig.add_subplot(gs[2, 1])\nsns.violinplot(data = df_punt, x='specialTeamsResult', y='hangTime')\n\nfig.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:42.869431Z","iopub.execute_input":"2021-10-29T08:15:42.869641Z","iopub.status.idle":"2021-10-29T08:15:44.584427Z","shell.execute_reply.started":"2021-10-29T08:15:42.869617Z","shell.execute_reply":"2021-10-29T08:15:44.583863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Просмотр результатов специальной команды и категориальных фичей\n#-------------------------------\n\n# 1. Построение таблицы значений результата специального действия и о том как пант что произошло с пантом\ndf_temp = pd.DataFrame(df_punt.groupby(['specialTeamsResult','kickContactType']).kickContactType.count().unstack().reset_index())\n#df2.columns = df2.columns.droplevel(0)\n#df2.columns = df2.columns.map(''.join)\ndf_temp = df_temp.fillna(0)\nprint(df_temp, 5*'\\n')\n\n# 2. Построение таблицы значений результата специального действия и о том была ли привязка у панта\ndf2_temp = pd.DataFrame(df_punt.groupby(['specialTeamsResult','snapDetail']).snapDetail.count().unstack().reset_index())\nprint(df2_temp, 5*'\\n')\n\n# 3.Построение таблицы значений результата специального действия и типа панта\ndf3_temp = pd.DataFrame(df_punt.groupby(['specialTeamsResult','kickType']).kickType.count().unstack().reset_index())\ndf3_temp = df3_temp.fillna(0)\nprint(df3_temp)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:44.585581Z","iopub.execute_input":"2021-10-29T08:15:44.585936Z","iopub.status.idle":"2021-10-29T08:15:44.616832Z","shell.execute_reply.started":"2021-10-29T08:15:44.585898Z","shell.execute_reply":"2021-10-29T08:15:44.61602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Исходя из скрипичных графиков характеристика operationTime практически одинакова для всех результатов Punt, поэтому не имеет смысла её учитывать.  \n#### Характеристика kickContactType может вызвать утечку данных, поскольку содержит типожирование исходов панта, а целью модели было выбрано предсказание самих исходов. Забегая вперед, при обучении модели добавление kickContactType повышало точность сразу на 15%, что и подверждает создание утечки данных.","metadata":{}},{"cell_type":"markdown","source":"## Обучение модели  ","metadata":{}},{"cell_type":"markdown","source":"#### Поставленная задача с выбранным набором данных относиться к задачи класификации. Посколько для каждого исхода есть набор характеристик, их сочетание и комбинации во время матча обеспечивают получение нужного исхода розыгрыша панта.","metadata":{}},{"cell_type":"code","source":"#Подготовка категориальных данных для модели\n#-------------------------------\n\nfrom sklearn.preprocessing import LabelEncoder\ndata2 = df_punt.copy()\nlabel_encoder = LabelEncoder()\n\ncolumns_LE = {\n    \"1\": 'kickContactType',\n    \"2\": 'snapDetail',\n    \"3\": 'kickType'}\n\nfor name, column in columns_LE.items():\n    print(data2[column].unique())\n    mapped_education = pd.Series(label_encoder.fit_transform(data2[column]))\n    data2[column] = label_encoder.fit_transform(data2[column])\n    print(dict(enumerate(label_encoder.classes_)))\n    print(data2[column].unique())\n\n#data2","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:44.617932Z","iopub.execute_input":"2021-10-29T08:15:44.618147Z","iopub.status.idle":"2021-10-29T08:15:44.64094Z","shell.execute_reply.started":"2021-10-29T08:15:44.618121Z","shell.execute_reply":"2021-10-29T08:15:44.640168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Выбор наиболее успешной модели на основе всех выбранных фичей\n#-------------------------------\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import confusion_matrix\n\n\nmodels = {\n    \"XGBClassifier\": XGBClassifier(),\n    \"K-Nearest Neighbors\": KNeighborsClassifier(),  \n    \"Stochastic Gradient Descent Classifier\": SGDClassifier(),\n    \"Support Vector Classifier\": SVC(),\n    \"Linear Support Vector Classifier\": LinearSVC(),\n    \"Decision Tree Classifier\": DecisionTreeClassifier(),\n    \"Random Forest Classifer\": RandomForestClassifier(random_state = 5)         \n         }\n \ncols_to_use2 = ['yardlineNumber', 'kickLength', 'playResult', 'hangTime', 'snapDetail', 'kickType']\nX4 = data2[cols_to_use2]\ny4 = data2.specialTeamsResult\nX4_train, X4_valid, y4_train, y4_valid = train_test_split(X4, y4, test_size=0.4, random_state = 11)\n\nfor name, model in models.items():\n    model.fit(X4_train, y4_train)\n    print(name + \" trained\")\n    \nprint(\"-------------------------\", '\\n')\n\nfor name, model in models.items():\n    print(name)\n    predictions4 = model.predict(X4_valid)\n    print(\"Accuracy: %.2f%%\" % (accuracy_score(y4_valid, predictions4, normalize=True) * 100.0))","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:44.641889Z","iopub.execute_input":"2021-10-29T08:15:44.642077Z","iopub.status.idle":"2021-10-29T08:15:47.692791Z","shell.execute_reply.started":"2021-10-29T08:15:44.642054Z","shell.execute_reply":"2021-10-29T08:15:47.692019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Модели XGBClassifier и RandomForestClassifier показали наибольшую точность. Попробуем подобрать для них параметры, позволяющие ещё больше повысить их точность","metadata":{}},{"cell_type":"code","source":"#Подбор наилучших параметров для модели XGBClassifier\n#-------------------------------\n\nfrom sklearn.model_selection import train_test_split\nX5 = data2[cols_to_use2]\ny5 = data2.specialTeamsResult\nX5_train, X5_valid, y5_train, y5_valid = train_test_split(X5, y5, test_size=0.4, random_state = 11)\n\nfrom xgboost import XGBClassifier\nmy_model5 = XGBClassifier(booster='gbtree', max_depth=7, eta=0.07, gamma=0.01, subsample=0.8, colsample_bytree = 1, min_child_weight=2)\nmy_model5.fit(X5_train, y5_train)\n\npredictions5 = my_model5.predict(X5_valid)\n\nfrom sklearn.metrics import accuracy_score\naccuracy = accuracy_score(y5_valid, predictions5, normalize=True)\nprint(\"Primary Accuracy: 70.57% (with standart parameters)\")\nprint(\"Accuracy: %.2f%%\" % (accuracy * 100.0))\n\nprint('\\n')\nprint(my_model5)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:47.694506Z","iopub.execute_input":"2021-10-29T08:15:47.69474Z","iopub.status.idle":"2021-10-29T08:15:48.842622Z","shell.execute_reply.started":"2021-10-29T08:15:47.694714Z","shell.execute_reply":"2021-10-29T08:15:48.841971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Подбор наилучших параметров для модели RandomForestClassifier\n#-------------------------------\n\nfrom sklearn.model_selection import train_test_split\nX6 = data2[cols_to_use2]\ny6 = data2.specialTeamsResult\nX6_train, X6_valid, y6_train, y6_valid = train_test_split(X6, y6, test_size=0.4, random_state = 11)\n\nfrom sklearn.ensemble import RandomForestClassifier\nmy_model6 = RandomForestClassifier(max_depth=15, n_estimators=500, max_features = 'auto', random_state = 10)\nmy_model6.fit(X6_train, y6_train)\n\npredictions6 = my_model6.predict(X6_valid)\n\nfrom sklearn.metrics import accuracy_score\naccuracy = accuracy_score(y6_valid, predictions6, normalize=True)\nprint(\"Primary Accuracy: 69.63% (with standart parameters)\")\nprint(\"Accuracy: %.2f%%\" % (accuracy * 100.0))\n\nprint('\\n')\nprint(my_model6)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T08:15:48.843611Z","iopub.execute_input":"2021-10-29T08:15:48.843827Z","iopub.status.idle":"2021-10-29T08:15:51.196169Z","shell.execute_reply.started":"2021-10-29T08:15:48.843802Z","shell.execute_reply":"2021-10-29T08:15:51.194531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Итог","metadata":{}},{"cell_type":"markdown","source":"### Модель XGBClassifier показывает наилучший результат и достигла точности в 71,97%.  \n### В дальнейшем планируется дополнить модель данными из трекинга игроков. Сочетание данных о положении мяча и игроков на поле после удара пантера должно существенно повысить качество модели и её преминимость в реальной игре на поле","metadata":{}}]}