{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"NFL Big Data Bowl 2022 ","metadata":{"id":"Nvba6voq3N8g"}},{"cell_type":"markdown","source":"Andrew Curtis Project","metadata":{"id":"nBiC0q1Z3by4"}},{"cell_type":"markdown","source":"Effect of weight and height of Kick Returners on Return Yardage","metadata":{"id":"sMQKPdLE3b_U"}},{"cell_type":"code","source":"!pip install pyspark\n!pip install -U -q PyDrive \n!apt install openjdk-8-jdk-headless -qq --yes\nimport os\nos.environ[\"JAVA_HOME\"] = \"/usr/lib/jvm/java-8-openjdk-amd64\"","metadata":{"id":"KF9_Fx59TRzD","execution":{"iopub.status.busy":"2022-03-02T20:27:52.111228Z","iopub.execute_input":"2022-03-02T20:27:52.111993Z","iopub.status.idle":"2022-03-02T20:29:07.597024Z","shell.execute_reply.started":"2022-03-02T20:27:52.111940Z","shell.execute_reply":"2022-03-02T20:29:07.595757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom pyspark import SparkContext, SparkFiles\nfrom pyspark.sql import SparkSession\nimport string\nimport matplotlib.pyplot as plt\nfrom pyspark.sql.functions import split\nfrom pyspark.sql.functions import col\nfrom pyspark.sql.types import IntegerType\n\nspark = SparkSession.builder.master(\"local\").appName(\"NFL\").getOrCreate()\n","metadata":{"id":"pbJwSg_4TmZr","execution":{"iopub.status.busy":"2022-03-02T20:29:07.601325Z","iopub.execute_input":"2022-03-02T20:29:07.601640Z","iopub.status.idle":"2022-03-02T20:29:13.186747Z","shell.execute_reply.started":"2022-03-02T20:29:07.601608Z","shell.execute_reply":"2022-03-02T20:29:13.185751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"playerData = spark.read.csv('../input/nfl-big-data-bowl-2022/players.csv', header = True)\nplayData = spark.read.csv('../input/nfl-big-data-bowl-2022/plays.csv', header = True)","metadata":{"id":"fb8iN_sAiJMD","execution":{"iopub.status.busy":"2022-03-02T20:29:13.188391Z","iopub.execute_input":"2022-03-02T20:29:13.188729Z","iopub.status.idle":"2022-03-02T20:29:19.479723Z","shell.execute_reply.started":"2022-03-02T20:29:13.188664Z","shell.execute_reply":"2022-03-02T20:29:19.478603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Pre-processing and Cleaning","metadata":{"id":"3FXjFavc3Jto"}},{"cell_type":"code","source":"#filter for only kick return plays and non-null returners\nreturnData = playData.filter(playData.kickReturnYardage != 'NA').filter(playData.returnerId != 'NA').drop(\"gameId\", \"playId\", \"quarter\", \"possessionTeam\", \"yardlineSide\", \"yardlineNumber\", \"gameClock\", \"penaltyJerseyNumbers\", \"preSnapHomeScore\", \"preSnapVisitorScore\", \"passResult\", \"absoluteYardlineNumber\")\n\nreducedPlayerData = playerData.drop(\"birthDate\", \"collegeName\", \"Position\", \"displayName\") #data cleaning\n\nreturnData = returnData.withColumnRenamed('returnerId', 'nflId') #to match player ID's\n\n#merge data sets\nmerged3 = returnData.join(reducedPlayerData, returnData.nflId == reducedPlayerData.nflId)\nmerged3 = merged3.filter(merged3.weight != 'NA').filter(merged3.height != 'NA')\nmerged3.show(5) ","metadata":{"id":"Zm7xAWFSlj05","outputId":"b6132919-360c-4993-c864-535f5a1b5f5e","execution":{"iopub.status.busy":"2022-03-02T20:29:19.482885Z","iopub.execute_input":"2022-03-02T20:29:19.483244Z","iopub.status.idle":"2022-03-02T20:29:21.025261Z","shell.execute_reply.started":"2022-03-02T20:29:19.483198Z","shell.execute_reply":"2022-03-02T20:29:21.024163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pre-processing for height and weight\nnew_height = merged3.withColumn(\"height_feet\", split(col(\"height\"), \"-\").getItem(0)).withColumn(\"height_inch\", split(col(\"height\"), \"-\").getItem(1))\nnew_height = new_height.withColumn(\"height_feet\", new_height[\"height_feet\"].cast(IntegerType()))\nnew_height = new_height.withColumn(\"height_inch\", new_height[\"height_inch\"].cast(IntegerType()))\nnew_height = new_height.withColumn(\"weight\", new_height[\"weight\"].cast(IntegerType()))\nnew_height = new_height.withColumn(\"kickReturnYardage\", new_height[\"kickReturnYardage\"].cast(IntegerType()))\n\nnew_height.show(5)","metadata":{"id":"FoleZc9fpJgg","outputId":"5bee0995-a424-48d5-dbeb-9450bc8552d0","execution":{"iopub.status.busy":"2022-03-02T20:29:21.026620Z","iopub.execute_input":"2022-03-02T20:29:21.026952Z","iopub.status.idle":"2022-03-02T20:29:21.877958Z","shell.execute_reply.started":"2022-03-02T20:29:21.026911Z","shell.execute_reply":"2022-03-02T20:29:21.877055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#more height and weight pre-processing\nnew_height = new_height.replace(4, 48, \"height_feet\")\nnew_height = new_height.replace(5, 60, \"height_feet\")\nnew_height = new_height.replace(6, 72, \"height_feet\")\nnew_height = new_height.replace(7, 84, \"height_feet\")\nnew_height = new_height.na.fill(value=0, subset=[\"height_inch\"])\n\n#final cleaned data for use\nfixedData = new_height.withColumn(\"totalHeight\", col(\"height_feet\")+col(\"height_inch\"))\nfixedData.show(10)\n","metadata":{"id":"ylCbpWlzuCmd","outputId":"44626bfe-81b9-4375-d947-cb1cb40cba3f","execution":{"iopub.status.busy":"2022-03-02T20:29:21.879198Z","iopub.execute_input":"2022-03-02T20:29:21.879605Z","iopub.status.idle":"2022-03-02T20:29:23.035333Z","shell.execute_reply.started":"2022-03-02T20:29:21.879560Z","shell.execute_reply":"2022-03-02T20:29:23.034724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#unused dataframe with just relavent data, for display purposes\n#returnYardsHeight = fixedData.drop(\"playDescription\", \"down\", \"yardsToGo\", \"specialTeamsPlayType\", \"specialTeamsResult\", \"kickerId\", \"nflId\", \"kickBlockerId\", \"penaltyCodes\", \"penaltyYards\", \"kickLength\", \"playResult\", \"height\", \"weight\", \"height_feet\", \"height_inch\")","metadata":{"id":"cXsUK_4tISez","execution":{"iopub.status.busy":"2022-03-02T20:29:23.036298Z","iopub.execute_input":"2022-03-02T20:29:23.036505Z","iopub.status.idle":"2022-03-02T20:29:23.041902Z","shell.execute_reply.started":"2022-03-02T20:29:23.036479Z","shell.execute_reply":"2022-03-02T20:29:23.040855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Linear Regression Models","metadata":{"id":"VvO5K6uv3EpM"}},{"cell_type":"code","source":"#multiple linear regression model\nfrom pyspark.ml.feature import VectorAssembler\nvectorAssembler = VectorAssembler(inputCols = ['weight', 'totalHeight'], outputCol = 'features')\nregression_df = vectorAssembler.transform(fixedData)\nregression_df = regression_df.select(['features', 'kickReturnYardage'])\nregression_df.show(3)","metadata":{"id":"TfaDpgQ8JTxh","outputId":"93ae31ef-b2a7-4f97-ff18-bb1f2d360bce","execution":{"iopub.status.busy":"2022-03-02T20:29:23.043924Z","iopub.execute_input":"2022-03-02T20:29:23.044531Z","iopub.status.idle":"2022-03-02T20:29:24.486630Z","shell.execute_reply.started":"2022-03-02T20:29:23.044481Z","shell.execute_reply":"2022-03-02T20:29:24.485662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiple linear regression results\nfrom pyspark.ml.regression import LinearRegression\nlr = LinearRegression(featuresCol = 'features', labelCol='kickReturnYardage')\nlr_model = lr.fit(regression_df)\nprint(\"Coefficients: \" + str(lr_model.coefficients))\nprint(\"Intercept: \" + str(lr_model.intercept))","metadata":{"id":"2S8YhRPxDe5K","outputId":"079da3c6-6214-4197-c98b-8aa1075f12fe","execution":{"iopub.status.busy":"2022-03-02T20:29:24.487896Z","iopub.execute_input":"2022-03-02T20:29:24.488774Z","iopub.status.idle":"2022-03-02T20:29:27.434847Z","shell.execute_reply.started":"2022-03-02T20:29:24.488728Z","shell.execute_reply":"2022-03-02T20:29:27.434182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#multiple linear regression R**2\ntrainingSummary = lr_model.summary\nprint(\"RMSE: %f\" % trainingSummary.rootMeanSquaredError)\nprint(\"r2: %f\" % trainingSummary.r2)","metadata":{"id":"9LcpbliLLaM8","outputId":"e06219b2-fdd5-425c-abff-685976d2601b","execution":{"iopub.status.busy":"2022-03-02T20:29:27.437570Z","iopub.execute_input":"2022-03-02T20:29:27.438041Z","iopub.status.idle":"2022-03-02T20:29:27.447056Z","shell.execute_reply.started":"2022-03-02T20:29:27.438008Z","shell.execute_reply":"2022-03-02T20:29:27.446133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#information regarding kick returns in data set\nregression_df.describe().show()","metadata":{"id":"5fPyzpGjLdm9","outputId":"58d8430b-9f31-4930-e1d3-eb3c8c3b85c3","execution":{"iopub.status.busy":"2022-03-02T20:29:27.448721Z","iopub.execute_input":"2022-03-02T20:29:27.449330Z","iopub.status.idle":"2022-03-02T20:29:28.353291Z","shell.execute_reply.started":"2022-03-02T20:29:27.449272Z","shell.execute_reply":"2022-03-02T20:29:28.352405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#simple linear regression with weight model\nfrom pyspark.ml.feature import VectorAssembler\nvectorAssembler = VectorAssembler(inputCols = ['weight'], outputCol = 'features2')\nregression_df2 = vectorAssembler.transform(fixedData)\nregression_df2 = regression_df2.select(['features2', 'kickReturnYardage'])\nregression_df2.show(3)","metadata":{"id":"gsfbxE7PPXLq","outputId":"ff3d7c3f-8c08-474e-a041-99dab37b34d9","execution":{"iopub.status.busy":"2022-03-02T20:29:28.354840Z","iopub.execute_input":"2022-03-02T20:29:28.355262Z","iopub.status.idle":"2022-03-02T20:29:28.841738Z","shell.execute_reply.started":"2022-03-02T20:29:28.355208Z","shell.execute_reply":"2022-03-02T20:29:28.840784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#weight linear regression results\nfrom pyspark.ml.regression import LinearRegression\nlr = LinearRegression(featuresCol = 'features2', labelCol='kickReturnYardage')\nlr_model = lr.fit(regression_df2)\nprint(\"Coefficients: \" + str(lr_model.coefficients))\nprint(\"Intercept: \" + str(lr_model.intercept))","metadata":{"id":"df5dzQZ0PfXL","outputId":"16666ae5-38d2-43ba-a66c-ae6c90a36127","execution":{"iopub.status.busy":"2022-03-02T20:29:28.842937Z","iopub.execute_input":"2022-03-02T20:29:28.843246Z","iopub.status.idle":"2022-03-02T20:29:30.165698Z","shell.execute_reply.started":"2022-03-02T20:29:28.843203Z","shell.execute_reply":"2022-03-02T20:29:30.164764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#weight linear regression r**2\ntrainingSummary = lr_model.summary\nprint(\"RMSE: %f\" % trainingSummary.rootMeanSquaredError)\nprint(\"r2: %f\" % trainingSummary.r2)","metadata":{"id":"38KrwWbmPkK0","outputId":"af0f5bf9-11f6-4f39-ec53-5e5075134aa7","execution":{"iopub.status.busy":"2022-03-02T20:29:30.167818Z","iopub.execute_input":"2022-03-02T20:29:30.168669Z","iopub.status.idle":"2022-03-02T20:29:30.177740Z","shell.execute_reply.started":"2022-03-02T20:29:30.168616Z","shell.execute_reply":"2022-03-02T20:29:30.177035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#simple linear regression using height model\nfrom pyspark.ml.feature import VectorAssembler\nvectorAssembler = VectorAssembler(inputCols = ['totalHeight'], outputCol = 'features3')\nregression_df3 = vectorAssembler.transform(fixedData)\nregression_df3 = regression_df3.select(['features3', 'kickReturnYardage'])\nregression_df3.show(3)","metadata":{"id":"bu660eFxP4r1","outputId":"4c100047-b077-4347-917a-de7aa17dc53f","execution":{"iopub.status.busy":"2022-03-02T20:29:30.179212Z","iopub.execute_input":"2022-03-02T20:29:30.179912Z","iopub.status.idle":"2022-03-02T20:29:30.837981Z","shell.execute_reply.started":"2022-03-02T20:29:30.179821Z","shell.execute_reply":"2022-03-02T20:29:30.836919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#height linear regression results\nfrom pyspark.ml.regression import LinearRegression\nlr = LinearRegression(featuresCol = 'features3', labelCol='kickReturnYardage')\nlr_model = lr.fit(regression_df3)\nprint(\"Coefficients: \" + str(lr_model.coefficients))\nprint(\"Intercept: \" + str(lr_model.intercept))","metadata":{"id":"7d9ozuKzP40W","outputId":"6c2632c1-1834-4c18-de6e-5b99321bae5c","execution":{"iopub.status.busy":"2022-03-02T20:29:30.839294Z","iopub.execute_input":"2022-03-02T20:29:30.839601Z","iopub.status.idle":"2022-03-02T20:29:32.159592Z","shell.execute_reply.started":"2022-03-02T20:29:30.839559Z","shell.execute_reply":"2022-03-02T20:29:32.158738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#height linear regression r**2\ntrainingSummary = lr_model.summary\nprint(\"RMSE: %f\" % trainingSummary.rootMeanSquaredError)\nprint(\"r2: %f\" % trainingSummary.r2)","metadata":{"id":"JFtlrFiaP47A","outputId":"9c3b2bb7-0602-47f1-ea22-c3d93ae4cb7a","execution":{"iopub.status.busy":"2022-03-02T20:29:32.162650Z","iopub.execute_input":"2022-03-02T20:29:32.163394Z","iopub.status.idle":"2022-03-02T20:29:32.171954Z","shell.execute_reply.started":"2022-03-02T20:29:32.163344Z","shell.execute_reply":"2022-03-02T20:29:32.171070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting","metadata":{"id":"kJZRCvWK29Sj"}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.stats import gaussian_kde\n\n#plot for distribution of weights\n\nx = np.array(fixedData.select('weight').collect())\nx = x[np.logical_not(np.isnan(x))]\n\ny = np.array(fixedData.select('kickReturnYardage').collect())\ny = y[np.logical_not(np.isnan(y))]\n\n# Calculate the point density\n# xy = np.vstack([x,y])\n# z = gaussian_kde(xy)(xy)\n\n# Sort the points by density, so that the densest points are plotted last\n# idx = z.argsort()\n# x, y, z = x[idx], y[idx], z[idx]\n\nplt.hist(x, bins =30, color = 'blue')\nplt.xlabel('Weight (lbs)', fontsize=16)\nplt.ylabel('counts', fontsize=16)\nplt.title('Distribution of Weights of Kick Returners', fontsize=16)\nplt.show()","metadata":{"id":"9YqO13gqV13K","outputId":"8adc4c0d-24dc-42cc-edd7-fae60bcdfb49","execution":{"iopub.status.busy":"2022-03-02T20:29:32.173472Z","iopub.execute_input":"2022-03-02T20:29:32.174158Z","iopub.status.idle":"2022-03-02T20:29:34.415196Z","shell.execute_reply.started":"2022-03-02T20:29:32.174078Z","shell.execute_reply":"2022-03-02T20:29:34.414250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot for distribution of heights\nx = np.array(fixedData.select('totalHeight').collect())\nx = x[np.logical_not(np.isnan(x))]\n\ny = np.array(fixedData.select('kickReturnYardage').collect())\ny = y[np.logical_not(np.isnan(y))]\n\n# Calculate the point density\n# xy = np.vstack([x,y])\n# z = gaussian_kde(xy)(xy)\n\n# Sort the points by density, so that the densest points are plotted last\n# idx = z.argsort()\n# x, y, z = x[idx], y[idx], z[idx]\n\nplt.hist(x, bins =30, color = 'blue')\nplt.xlabel('Height (inches)', fontsize=16)\nplt.ylabel('counts', fontsize=16)\nplt.title('Distribution of Heights of Kick Returners', fontsize=16)\nplt.show()","metadata":{"id":"ksmPFLcSis17","outputId":"89099155-1dec-49d6-cfa1-0a6c726cf82f","execution":{"iopub.status.busy":"2022-03-02T20:29:34.416966Z","iopub.execute_input":"2022-03-02T20:29:34.417493Z","iopub.status.idle":"2022-03-02T20:29:35.681888Z","shell.execute_reply.started":"2022-03-02T20:29:34.417444Z","shell.execute_reply":"2022-03-02T20:29:35.680991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot for distribution of kick return yardages\nx = np.array(fixedData.select('weight').collect())\nx = x[np.logical_not(np.isnan(x))]\n\ny = np.array(fixedData.select('kickReturnYardage').collect())\ny = y[np.logical_not(np.isnan(y))]\n\n# Calculate the point density\n# xy = np.vstack([x,y])\n# z = gaussian_kde(xy)(xy)\n\n# Sort the points by density, so that the densest points are plotted last\n# idx = z.argsort()\n# x, y, z = x[idx], y[idx], z[idx]\n\nplt.hist(y, bins =30, color = 'blue')\nplt.xlabel('Kick Return Yardage', fontsize=16)\nplt.ylabel('counts', fontsize=16)\nplt.title('Distribution of Kick Return Yardages', fontsize=16)\nplt.show()","metadata":{"id":"SbA6mAbtfapb","outputId":"1805aa6d-4eab-42ec-fc6c-fe069db2c3fe","execution":{"iopub.status.busy":"2022-03-02T20:29:35.683207Z","iopub.execute_input":"2022-03-02T20:29:35.683502Z","iopub.status.idle":"2022-03-02T20:29:36.830964Z","shell.execute_reply.started":"2022-03-02T20:29:35.683468Z","shell.execute_reply":"2022-03-02T20:29:36.829756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#density plot of kick return yardage vs weight\nx = np.array(fixedData.select('weight').collect())\nx = x[np.logical_not(np.isnan(x))]\n\ny = np.array(fixedData.select('kickReturnYardage').collect())\ny = y[np.logical_not(np.isnan(y))]\n\n# Calculate the point density\nxy = np.vstack([x,y])\nz = gaussian_kde(xy)(xy)\n\n# Sort the points by density, so that the densest points are plotted last\nidx = z.argsort()\nx, y, z = x[idx], y[idx], z[idx]\n\nfig, ax = plt.subplots()\nax.scatter(x, y, c=z, s=50)\nplt.xlabel('Weight (lbs)', fontsize=16)\nplt.ylabel('Kick Return Yardage', fontsize=16)\nplt.title('Kick Return Yardage vs. Weight', fontsize=16)\nplt.show()","metadata":{"id":"ZZcTUQvYeMiP","outputId":"3d5143f9-b218-485a-a6d3-8dce227e22ad","execution":{"iopub.status.busy":"2022-03-02T20:29:36.832491Z","iopub.execute_input":"2022-03-02T20:29:36.832946Z","iopub.status.idle":"2022-03-02T20:29:38.221569Z","shell.execute_reply.started":"2022-03-02T20:29:36.832896Z","shell.execute_reply":"2022-03-02T20:29:38.220694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#density plot of kick return yardage vs height\nx = np.array(fixedData.select('totalHeight').collect())\nx = x[np.logical_not(np.isnan(x))]\n\ny = np.array(fixedData.select('kickReturnYardage').collect())\ny = y[np.logical_not(np.isnan(y))]\n\n# Calculate the point density\nxy = np.vstack([x,y])\nz = gaussian_kde(xy)(xy)\n\n# Sort the points by density, so that the densest points are plotted last\nidx = z.argsort()\nx, y, z = x[idx], y[idx], z[idx]\n\nfig, ax = plt.subplots()\nax.scatter(x, y, c=z, s=50)\nplt.xlabel('Height (inches)', fontsize=16)\nplt.ylabel('Kick Return Yardage', fontsize=16)\nplt.title('Kick Return Yardage vs. height', fontsize=16)\nplt.show()","metadata":{"id":"Er-boLeRkDL8","outputId":"8f7ec81e-e744-472d-a14c-8beed8cff80b","execution":{"iopub.status.busy":"2022-03-02T20:29:38.222969Z","iopub.execute_input":"2022-03-02T20:29:38.223384Z","iopub.status.idle":"2022-03-02T20:29:39.598337Z","shell.execute_reply.started":"2022-03-02T20:29:38.223337Z","shell.execute_reply":"2022-03-02T20:29:39.595940Z"},"trusted":true},"execution_count":null,"outputs":[]}]}