{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-09T09:05:11.255379Z","iopub.execute_input":"2022-01-09T09:05:11.256554Z","iopub.status.idle":"2022-01-09T09:05:11.279551Z","shell.execute_reply.started":"2022-01-09T09:05:11.256430Z","shell.execute_reply":"2022-01-09T09:05:11.278694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PART 1: Import Data and Dataframe Creation","metadata":{}},{"cell_type":"code","source":"from sklearn.datasets import fetch_20newsgroups\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nplays_raw_data = pd.read_csv(\"../input/nfl-big-data-bowl-2022/plays.csv\")\nplays = plays_raw_data\n\n# type - results\nspecialTeamsPlay = plays.groupby(['specialTeamsResult', 'specialTeamsPlayType']).size().unstack()\nspecialTeamsPlay","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:11.281618Z","iopub.execute_input":"2022-01-09T09:05:11.282208Z","iopub.status.idle":"2022-01-09T09:05:12.731153Z","shell.execute_reply.started":"2022-01-09T09:05:11.282147Z","shell.execute_reply":"2022-01-09T09:05:12.730005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plays_Punt = plays.loc[plays.specialTeamsPlayType=='Punt'].groupby(['specialTeamsResult', 'specialTeamsPlayType']).size().unstack()\n# plays_Kickoff = plays.loc[plays.specialTeamsPlayType=='Kickoff'].groupby(['specialTeamsResult', 'specialTeamsPlayType']).size().unstack()\n# plays_FieldGoal = plays.loc[plays.specialTeamsPlayType=='Field Goal'].groupby(['specialTeamsResult', 'specialTeamsPlayType']).size().unstack()\n# plays_ExtraPoint = plays.loc[plays.specialTeamsPlayType=='Extra Point'].groupby(['specialTeamsResult', 'specialTeamsPlayType']).size().unstack()\n\n# plays_Punt_Result = ['Blocked Punt', 'Downed', 'Fair Catch','Muffed', 'Non-Special Teams Result', 'Out of Bounds', 'Return', 'Touchback']\n# plays_Punt","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.733036Z","iopub.execute_input":"2022-01-09T09:05:12.733312Z","iopub.status.idle":"2022-01-09T09:05:12.739875Z","shell.execute_reply.started":"2022-01-09T09:05:12.733269Z","shell.execute_reply":"2022-01-09T09:05:12.738627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plays_Kickoff_Result = ['Downed', 'Fair Catch', 'Kickoff Team Recovery', 'Muffed', 'Out of Bounds', 'Return', 'Touchback']\n# plays_Kickoff","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.741299Z","iopub.execute_input":"2022-01-09T09:05:12.741593Z","iopub.status.idle":"2022-01-09T09:05:12.755092Z","shell.execute_reply.started":"2022-01-09T09:05:12.741563Z","shell.execute_reply":"2022-01-09T09:05:12.754233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plays_FieldGoal_Result = ['Blocked Kick Attempt', 'Downed', 'Kick Attempt Good', 'Kick Attempt No Good', 'Non-Special Teams Result', 'Out of Bounds']\n# plays_FieldGoal","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.757770Z","iopub.execute_input":"2022-01-09T09:05:12.758020Z","iopub.status.idle":"2022-01-09T09:05:12.777121Z","shell.execute_reply.started":"2022-01-09T09:05:12.757990Z","shell.execute_reply":"2022-01-09T09:05:12.775969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plays_ExtraPoint_Result = ['Blocked Kick Attempt', 'Kick Attempt Good', 'Kick Attempt No Good', 'Non-Special Teams Result']\n# plays_ExtraPoint","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.778797Z","iopub.execute_input":"2022-01-09T09:05:12.779205Z","iopub.status.idle":"2022-01-09T09:05:12.790020Z","shell.execute_reply.started":"2022-01-09T09:05:12.779164Z","shell.execute_reply":"2022-01-09T09:05:12.789347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# specialTeamsResultList = ['Blocked Kick Attempt', 'Blocked Punt', 'Downed', \n#                           'Fair Catch', 'Kick Attempt Good', 'Kick Attempt No Good', \n#                           'Kickoff Team Recovery', 'Muffed', 'Non-Special Teams Result', \n#                           'Out of Bounds', 'Return', 'Touchback']\n# plt.figure(figsize=(8,8))\n# plt.pie(plays_ExtraPoint['Extra Point'], explode=(0.5,0.3,0.1,0.5), autopct='%1.1f%%' )\n# plt.legend(loc=7,  labels = plays_ExtraPoint_Result, title = \"Extra Point:\", bbox_to_anchor=(0.6, 0.7))\n# plt.show()\n\n# plt.figure(figsize=(8,8))\n# plt.pie(plays_FieldGoal['Field Goal'], explode=(0.5, 0.5,0.1,0.1,0.5,0.5), autopct='%1.1f%%')\n# plt.legend(loc=7,  labels = plays_FieldGoal_Result, title = \"Field Goal:\", bbox_to_anchor=(0.85, 0.8))\n# plt.show()\n\n# plt.figure(figsize=(8,8))\n# plt.pie(plays_Kickoff['Kickoff'], explode=(0.5,0.5,0.5,0.5,0.5,0.1,0.1), autopct='%1.1f%%')\n# plt.legend(loc=7,  labels = plays_Kickoff_Result, title = \"Kickoff:\", bbox_to_anchor=(1, 0.2))\n# plt.show()\n\n# plt.figure(figsize=(8,8))\n# plt.pie(plays_Punt['Punt'], explode=(0.1,0.1,0.1,0.1,0.1,0.1,0.1,0.1), autopct='%1.1f%%' )\n# plt.legend(loc=7,  labels = plays_Punt_Result, title = \"Punt:\", bbox_to_anchor=(1.5, 0.5))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.791106Z","iopub.execute_input":"2022-01-09T09:05:12.791466Z","iopub.status.idle":"2022-01-09T09:05:12.805526Z","shell.execute_reply.started":"2022-01-09T09:05:12.791388Z","shell.execute_reply":"2022-01-09T09:05:12.804421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create a dataframe that shows the frequency play results with the involvement of each player in kick","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\n# player - results\nkickers_results = plays.groupby(['kickerId','specialTeamsResult']).size().unstack().replace(np.nan, 0).reset_index()\nkickers_results","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.808004Z","iopub.execute_input":"2022-01-09T09:05:12.808601Z","iopub.status.idle":"2022-01-09T09:05:12.858859Z","shell.execute_reply.started":"2022-01-09T09:05:12.808552Z","shell.execute_reply":"2022-01-09T09:05:12.857977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create a dataframe that shows the frequency play results with the involvement of each player in return","metadata":{}},{"cell_type":"code","source":"returners_results = plays.groupby(['returnerId','specialTeamsResult']).size().unstack().replace(np.nan, 0).reset_index()\nreturners_results","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.861714Z","iopub.execute_input":"2022-01-09T09:05:12.862016Z","iopub.status.idle":"2022-01-09T09:05:12.897422Z","shell.execute_reply.started":"2022-01-09T09:05:12.861987Z","shell.execute_reply":"2022-01-09T09:05:12.896324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read the players dataset and define a function that query players name","metadata":{}},{"cell_type":"code","source":"players_raw_data = pd.read_csv(\"../input/nfl-big-data-bowl-2022/players.csv\")\nplayers = players_raw_data\n\ndef queryName(i):\n    name = players.query('nflId == @i').displayName.values\n    return name ","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.898850Z","iopub.execute_input":"2022-01-09T09:05:12.899076Z","iopub.status.idle":"2022-01-09T09:05:12.921860Z","shell.execute_reply.started":"2022-01-09T09:05:12.899047Z","shell.execute_reply":"2022-01-09T09:05:12.920668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Getting the returners' names","metadata":{}},{"cell_type":"code","source":"X=plays\nIDList = X.returnerId\nNameList = []\nfor i in IDList.values:\n    try:\n        if isinstance(i, float):\n            NameList.append(np.nan)\n        else:\n            NameList.append(queryName(int(float(i)))[0])\n    except:\n        NameList.append(np.nan)\n\nX['returner_name'] = NameList\nreturnerName_results = X.groupby(['returner_name','specialTeamsResult']).size().unstack().replace(np.nan, 0).reset_index()\nreturnerName_results","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:12.925357Z","iopub.execute_input":"2022-01-09T09:05:12.926836Z","iopub.status.idle":"2022-01-09T09:05:27.729775Z","shell.execute_reply.started":"2022-01-09T09:05:12.926770Z","shell.execute_reply":"2022-01-09T09:05:27.728042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Getting the kickers' names","metadata":{}},{"cell_type":"code","source":"IDList = X.kickerId\nNameList = []\nfor i in IDList.values:\n    try:\n        if isinstance(i, int):\n            NameList.append(np.nan)\n        else:\n            NameList.append(queryName(i)[0])\n    except:\n        NameList.append(np.nan)\nX['kicker_name'] = NameList\nkickerName_results = X.groupby(['kicker_name','specialTeamsResult']).size().unstack().replace(np.nan, 0).reset_index()\nkickerName_results","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:05:27.732125Z","iopub.execute_input":"2022-01-09T09:05:27.732397Z","iopub.status.idle":"2022-01-09T09:06:10.152536Z","shell.execute_reply.started":"2022-01-09T09:05:27.732366Z","shell.execute_reply":"2022-01-09T09:06:10.151283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remove duplicates","metadata":{}},{"cell_type":"code","source":"# Drop these two rows because of duplicated names:\nkickerName_results = kickerName_results[kickerName_results.kicker_name != \"Aaron Brewer\"]\nkickerName_results = kickerName_results[kickerName_results.kicker_name != \"Chris Jones\"]\nkickerName_results = kickerName_results.reset_index(drop=True)\nkickerName_results","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:06:10.154534Z","iopub.execute_input":"2022-01-09T09:06:10.154869Z","iopub.status.idle":"2022-01-09T09:06:10.191987Z","shell.execute_reply.started":"2022-01-09T09:06:10.154825Z","shell.execute_reply":"2022-01-09T09:06:10.191135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get the BMI and position for kicker dataframe","metadata":{}},{"cell_type":"code","source":"def queryHeight(i):\n    height = players.query('`displayName` == @i').height.values\n    return height\ndef queryWeight(i):\n    weight = players.query('`displayName` == @i').weight.values\n    return weight\ndef feetToM(f,i):\n    i += f * 12\n    return round(i * 2.54, 1)/100\ndef queryPosition(i):\n    position = players.query('`displayName` == @i').Position.values\n    return position\ndef getPositionIndex(i):\n    if i == \"WR\":\n        return 0\n    elif i == \"RB\":\n        return 1\n    elif i == \"CB\":\n        return 2\n    elif i == \"TE\":\n        return 3\n    elif i == \"FB\":\n        return 4\n    elif i == \"FS\":\n        return 5\n    elif i == \"SS\":\n        return 6\n    elif i == \"ILB\":\n        return 7\n    elif i == \"OLB\":\n        return 8\n    elif i == \"MLB\":\n        return 9\n    elif i == \"QB\":\n        return 10\n    elif i == \"NT\":\n        return 11\n    elif i == \"LB\":\n        return 12\n    elif i == \"DE\":\n        return 13\n    elif i == \"G\":\n        return 14\n    elif i == \"DB\":\n        return 15\n    elif i == \"P\":\n        return 16\n    elif i == \"K\":\n        return 17\n    else:\n        return 20\n\nNameList = kickerName_results.kicker_name\nHeightList = []\nHeightList2 = []\nWeightList = []\nBMI = []\nPositionList = []\n\n# Obtain H and W \nfor i in NameList.values:\n    try:\n        HeightList.append(queryHeight(i)[0])\n        WeightList.append(queryWeight(i)[0])\n    except:\n        HeightList.append(np.nan)\n        WeightList.append(np.nan)\n\n# Change height scale\nfor i in HeightList:\n    try: \n        feet = i.split('-')[0]\n        inch = i.split('-')[1]\n        m = feetToM(float(feet),float(inch))\n        HeightList2.append(m)\n    except:\n        feet = [char for char in i][0]\n        inch = [char for char in i][1]\n        m = feetToM(float(feet),float(inch))\n        HeightList2.append(m)    \n\n# Compute BMI\nfor h, w in zip(HeightList2, WeightList):\n    BMI.append(round(w * 0.453592 / (h * h), 2))\n    \n# Obtain Position\nfor i in NameList.values:\n    try:\n        PositionList.append(getPositionIndex(queryPosition(i)[0]))\n    except:\n        PositionList.append(np.nan)\n        \nkickerName_results2 = kickerName_results.copy(deep=True)\nkickerName_results2['BMI'] = BMI\nkickerName_results2['Position'] = PositionList\nkickerName_results2","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:06:10.193498Z","iopub.execute_input":"2022-01-09T09:06:10.193727Z","iopub.status.idle":"2022-01-09T09:06:10.928373Z","shell.execute_reply.started":"2022-01-09T09:06:10.193698Z","shell.execute_reply":"2022-01-09T09:06:10.927378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get the BMI and position for returner dataframe","metadata":{}},{"cell_type":"code","source":"IDList = X.returnerId\n    \nNameList = returnerName_results.returner_name\nHeightList = []\nHeightList2 = []\nWeightList = []\nBMI = []\nPositionList = []\n\n# Obtain H and W \nfor i in NameList.values:\n    try:\n        HeightList.append(queryHeight(i)[0])\n        WeightList.append(queryWeight(i)[0])\n    except:\n        HeightList.append(np.nan)\n        WeightList.append(np.nan)\n\n# Change height scale\nfor i in HeightList:\n    try: \n        feet = i.split('-')[0]\n        inch = i.split('-')[1]\n        m = feetToM(float(feet),float(inch))\n        HeightList2.append(m)\n    except:\n        feet = [char for char in i][0]\n        inch = [char for char in i][1]\n        m = feetToM(float(feet),float(inch))\n        HeightList2.append(m)    \n\n# Compute BMI\nfor h, w in zip(HeightList2, WeightList):\n    BMI.append(round(w * 0.453592 / (h * h), 2))\n    \n# Obtain Position\nfor i in NameList.values:\n    try:\n        PositionList.append(getPositionIndex(queryPosition(i)[0]))\n    except:\n        PositionList.append(np.nan)\n#         print('haha')\n      \nreturnerName_results2 = returnerName_results.copy(deep=True)\nreturnerName_results2['BMI'] = BMI\nreturnerName_results2['Position'] = PositionList\nreturnerName_results2","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:06:10.930322Z","iopub.execute_input":"2022-01-09T09:06:10.930949Z","iopub.status.idle":"2022-01-09T09:06:13.254329Z","shell.execute_reply.started":"2022-01-09T09:06:10.930902Z","shell.execute_reply":"2022-01-09T09:06:13.253301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import seaborn as sns\n# plt.figure(figsize=(15,30))\n# ax = sns.heatmap(kickerName_results,\n#                  cmap=\"PuRd\",\n#                  vmin=0, vmax=200, annot=True)\n\n# returnerName_results2['Position'].value_counts()\n# a = returnerName_results2\n# a['Position'].replace({\"WR\": 0, \"b\": \"y\"}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:06:13.255892Z","iopub.execute_input":"2022-01-09T09:06:13.256139Z","iopub.status.idle":"2022-01-09T09:06:13.261118Z","shell.execute_reply.started":"2022-01-09T09:06:13.256110Z","shell.execute_reply":"2022-01-09T09:06:13.259933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PART 2: CLUSTERING - RETURNER","metadata":{}},{"cell_type":"markdown","source":"Clustering: find the pattern of returner's information and play results. Preprocessing:","metadata":{}},{"cell_type":"code","source":"# clustering preprocessing \nfrom sklearn.preprocessing import MinMaxScaler\n\n# Instantiate the object\nreturnerName_results3 = returnerName_results2.drop('returner_name', 1)\n\nscaler = MinMaxScaler()\n# Fit and transform the data\n# StandardScaler()\nreturnerName_results3 = scaler.fit_transform(returnerName_results3)\nreturnerName_results3","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:52:13.281215Z","iopub.execute_input":"2022-01-09T09:52:13.282011Z","iopub.status.idle":"2022-01-09T09:52:13.304444Z","shell.execute_reply.started":"2022-01-09T09:52:13.281943Z","shell.execute_reply":"2022-01-09T09:52:13.303336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Find the optimum number of clusters","metadata":{}},{"cell_type":"code","source":"returnerName_results3.shape","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:52:14.796319Z","iopub.execute_input":"2022-01-09T09:52:14.797612Z","iopub.status.idle":"2022-01-09T09:52:14.805779Z","shell.execute_reply.started":"2022-01-09T09:52:14.797512Z","shell.execute_reply":"2022-01-09T09:52:14.804851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the library\nimport seaborn as sns\nfrom sklearn.cluster import KMeans\nnp.random.seed(42)\ninertia = []\n\nstart = 1\nend = 20\n\n# Iterating the process\nfor i in range(start, end):\n  # Instantiate the model\n    model = KMeans(n_clusters=i)\n  # Fit The Model\n    model.fit(returnerName_results3)\n  # Extract the error of the model\n    inertia.append(model.inertia_)# Visualize the model\nsns.pointplot(x=list(range(start, end)), y=inertia)\nplt.title('SSE on K-Means based on # of clusters')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:52:16.656390Z","iopub.execute_input":"2022-01-09T09:52:16.657509Z","iopub.status.idle":"2022-01-09T09:52:18.228370Z","shell.execute_reply.started":"2022-01-09T09:52:16.657453Z","shell.execute_reply":"2022-01-09T09:52:18.227503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"clustering","metadata":{}},{"cell_type":"code","source":"np.random.seed(42)\n\n# Instantiate the model\nresults_returner = returnerName_results2.copy(deep=True)\nmodel = KMeans(n_clusters=4)\n# Fit the model\nmodel.fit(returnerName_results3)\n# Predict the cluster from the data and save it\ncluster = model.predict(returnerName_results3)\n# Add to the dataframe and show the result\nresults_returner['cluster'] = cluster\nresults_returner","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:52:20.382458Z","iopub.execute_input":"2022-01-09T09:52:20.382755Z","iopub.status.idle":"2022-01-09T09:52:20.450103Z","shell.execute_reply.started":"2022-01-09T09:52:20.382723Z","shell.execute_reply":"2022-01-09T09:52:20.449502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"visualisation:","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Create the dataframe to ease our visualization process\nvisualize = pd.DataFrame(model.cluster_centers_) #.reset_index()\nvisualize = visualize.T\nvisualize['column'] = ['Return', 'Fair Catch', 'Muffed', 'Kick Attempt No Good', 'BMI', 'Position']\nvisualize = visualize.melt(id_vars=['column'], var_name='cluster')\nvisualize['cluster'] = visualize.cluster.astype('category')\n# Visualize the result\nplt.figure(figsize=(12, 8))\nsns.barplot(x='cluster', y='value', hue='column', data=visualize)\nplt.title('The cluster\\'s characteristics')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:53:04.186745Z","iopub.execute_input":"2022-01-09T09:53:04.187061Z","iopub.status.idle":"2022-01-09T09:53:04.851114Z","shell.execute_reply.started":"2022-01-09T09:53:04.187030Z","shell.execute_reply":"2022-01-09T09:53:04.850124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:10:27.263586Z","iopub.execute_input":"2022-01-09T09:10:27.264106Z","iopub.status.idle":"2022-01-09T09:10:27.284799Z","shell.execute_reply.started":"2022-01-09T09:10:27.264065Z","shell.execute_reply":"2022-01-09T09:10:27.283553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PART 3: CLUSTERING - KICKER","metadata":{}},{"cell_type":"code","source":"# Instantiate the object\nkickerName_results3 = kickerName_results2.drop('kicker_name', 1)\nkickerName_results3 = kickerName_results3.drop('Blocked Punt', 1)\nkickerName_results3 = kickerName_results3.drop('Kickoff Team Recovery', 1)\nkickerName_results3 = kickerName_results3.drop('Blocked Kick Attempt', 1)\nkickerName_results3 = kickerName_results3.drop('Touchback', 1)\nkickerName_results3 = kickerName_results3.drop('Downed', 1)\nkickerName_results3 = kickerName_results3.drop('Muffed', 1)\n\ncolumn = 'BMI'\nkickerName_results3[column] = kickerName_results3[column] /kickerName_results3[column].abs().max()\ncolumn = 'Position'\nkickerName_results3[column] = kickerName_results3[column] /kickerName_results3[column].abs().max()\n\n\nscaler = MinMaxScaler()\n# Fit and transform the data\n# StandardScaler()\nkickerName_results3 = scaler.fit_transform(kickerName_results3)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:33:05.185313Z","iopub.execute_input":"2022-01-09T09:33:05.185817Z","iopub.status.idle":"2022-01-09T09:33:05.203998Z","shell.execute_reply.started":"2022-01-09T09:33:05.185766Z","shell.execute_reply":"2022-01-09T09:33:05.203276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kickerName_results3.shape","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:33:06.505287Z","iopub.execute_input":"2022-01-09T09:33:06.505809Z","iopub.status.idle":"2022-01-09T09:33:06.512003Z","shell.execute_reply.started":"2022-01-09T09:33:06.505757Z","shell.execute_reply":"2022-01-09T09:33:06.511284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\ninertia = []\n\nstart = 1\nend = 20\n\n# Iterating the process\nfor i in range(start, end):\n  # Instantiate the model\n    model = KMeans(n_clusters=i)\n  # Fit The Model\n    model.fit(kickerName_results3)\n  # Extract the error of the model\n    inertia.append(model.inertia_)# Visualize the model\nsns.pointplot(x=list(range(start, end)), y=inertia)\nplt.title('SSE on K-Means based on # of clusters')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:33:07.248548Z","iopub.execute_input":"2022-01-09T09:33:07.248834Z","iopub.status.idle":"2022-01-09T09:33:08.602069Z","shell.execute_reply.started":"2022-01-09T09:33:07.248802Z","shell.execute_reply":"2022-01-09T09:33:08.601461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\n\n# Instantiate the model\nresults_kicker = kickerName_results2.copy(deep=True)\nmodel = KMeans(n_clusters=3)\n# Fit the model\nmodel.fit(kickerName_results3)\n# Predict the cluster from the data and save it\ncluster = model.predict(kickerName_results3)\n# Add to the dataframe and show the result\nresults_kicker['cluster'] = cluster\nresults_kicker","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:34:09.666549Z","iopub.execute_input":"2022-01-09T09:34:09.667374Z","iopub.status.idle":"2022-01-09T09:34:09.736440Z","shell.execute_reply.started":"2022-01-09T09:34:09.667334Z","shell.execute_reply":"2022-01-09T09:34:09.735263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Create the dataframe to ease our visualization process\nvisualize = pd.DataFrame(model.cluster_centers_) #.reset_index()\n\n\nvisualize = visualize.T\n# visualize['column'] = ['Downed', 'Fair Catch', 'Muffed', 'Out of Bounds', 'Return', 'Touchback', 'Blocked Kick Attempt', 'Kick Attempt Good', 'Kick Attempt No Good', 'Kickoff Team Recovery', 'Blocked Punt', 'BMI', 'Position']\nvisualize['column'] = ['Fair Catch', 'Out of Bounds', 'Return', 'Kick Attempt Good', 'Kick Attempt No Good', 'BMI', 'Position']\nvisualize = visualize.melt(id_vars=['column'], var_name='cluster')\nvisualize['cluster'] = visualize.cluster.astype('category')\n# Visualize the result\nplt.figure(figsize=(12, 8))\nsns.barplot(x='cluster', y='value', hue='column', data=visualize)\nplt.title('The cluster\\'s characteristics')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-09T09:34:11.793058Z","iopub.execute_input":"2022-01-09T09:34:11.793891Z","iopub.status.idle":"2022-01-09T09:34:12.184926Z","shell.execute_reply.started":"2022-01-09T09:34:11.793850Z","shell.execute_reply":"2022-01-09T09:34:12.183958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}