{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-04T06:51:55.562315Z","iopub.execute_input":"2022-10-04T06:51:55.562710Z","iopub.status.idle":"2022-10-04T06:51:56.703694Z","shell.execute_reply.started":"2022-10-04T06:51:55.562632Z","shell.execute_reply":"2022-10-04T06:51:56.702146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XY position of player p0 before team A goals\nLet's first look at player p0 (who is on team A) within 10 seconds of team A scoring a goal.","metadata":{}},{"cell_type":"code","source":"# Loading in the data using parquet files prepared by reybahl\n# Parquet files have an advantage over the original csv files in that\n# you can choose specific columns to load to memory\n\nparquet_folder = '/kaggle/input/tps-oct-2022-compressed-parquet-files'\nd = {\"t0\":None,\"t1\":None,\"t2\":None,\"t3\":None,\"t4\":None,\n     \"t5\":None,\"t6\":None,\"t7\":None,\"t8\":None,\"t9\":None}\nfor i, k in enumerate(d):\n    d[k] = pd.read_parquet(os.path.join(parquet_folder,\n                                        f\"train_{i}.parquet.gzip\"),\n                           columns=['p0_pos_x', 'p0_pos_y',\n                                    'p1_pos_x', 'p1_pos_y',\n                                    'p2_pos_x', 'p2_pos_y',\n                                    'team_A_scoring_within_10sec'])\n\ndf = pd.concat([d['t0'], d['t1'], d['t2'], d['t3'], d['t4'],\n                d['t5'], d['t6'], d['t7'], d['t8'], d['t9']]).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:08:45.905702Z","iopub.execute_input":"2022-10-04T08:08:45.906956Z","iopub.status.idle":"2022-10-04T08:08:49.628789Z","shell.execute_reply.started":"2022-10-04T08:08:45.906902Z","shell.execute_reply":"2022-10-04T08:08:49.626918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:08:52.815205Z","iopub.execute_input":"2022-10-04T08:08:52.815609Z","iopub.status.idle":"2022-10-04T08:08:52.836795Z","shell.execute_reply.started":"2022-10-04T08:08:52.815576Z","shell.execute_reply":"2022-10-04T08:08:52.835120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I divide the x axis into 83 chunks and y axis into 240 chunks so that I can visualize the xy plane of the arena as a grid.","metadata":{}},{"cell_type":"code","source":"# [(-83, -81), (-81, -79), ..., (-3, -1), (-1, 1), ... (81, 83)]\nx_intervals = [(2*x-3, 2*x-1) for x in range(-40, 43)]\nx_bins = pd.IntervalIndex.from_tuples(x_intervals)\n# [(-120, -118), (-118, -116) ... (-2, 0), (0, 2), ... (116, 118), (118, 120)]\ny_intervals = [(2*y-2, 2*y) for y in range(-59, 61)] \ny_bins = pd.IntervalIndex.from_tuples(y_intervals)\n\ndf['p0_x'] = pd.cut(df.p0_pos_x, bins=x_bins)\ndf['p0_y'] = pd.cut(df.p0_pos_y, bins=y_bins)","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:09:35.332222Z","iopub.execute_input":"2022-10-04T08:09:35.332683Z","iopub.status.idle":"2022-10-04T08:10:18.564765Z","shell.execute_reply.started":"2022-10-04T08:09:35.332639Z","shell.execute_reply":"2022-10-04T08:10:18.564074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then I count the number of times team A scores when player p0 is in each xy grid cell.","metadata":{}},{"cell_type":"code","source":"p0_A = pd.DataFrame(df.groupby(['p0_x', 'p0_y'])\n                    ['team_A_scoring_within_10sec'].sum()).reset_index()\np0_A_grid = p0_A.pivot(index='p0_y',\n                       columns='p0_x',\n                       values='team_A_scoring_within_10sec')","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:10:18.566161Z","iopub.execute_input":"2022-10-04T08:10:18.566749Z","iopub.status.idle":"2022-10-04T08:10:19.302879Z","shell.execute_reply.started":"2022-10-04T08:10:18.566722Z","shell.execute_reply":"2022-10-04T08:10:19.301198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p0_A_grid","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:10:19.304350Z","iopub.execute_input":"2022-10-04T08:10:19.304711Z","iopub.status.idle":"2022-10-04T08:10:19.355052Z","shell.execute_reply.started":"2022-10-04T08:10:19.304679Z","shell.execute_reply":"2022-10-04T08:10:19.353554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Displaying this data as as heatmap, it looks like this.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 12))\nsns.heatmap(p0_A_grid)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:10:34.255857Z","iopub.execute_input":"2022-10-04T08:10:34.257057Z","iopub.status.idle":"2022-10-04T08:10:35.217120Z","shell.execute_reply.started":"2022-10-04T08:10:34.256954Z","shell.execute_reply":"2022-10-04T08:10:35.215301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"However, this heatmap does not take into account how often player p0 is in each grid cell in general, regardless of team A scoring. So we see starting positions light up the brightest. In order to adjust for this, we divide by the number of times each xy position is occupied by player p0 in the entire game regardless of team A scoring or not.","metadata":{}},{"cell_type":"code","source":"p0_A_counts = pd.DataFrame(df[['p0_x', 'p0_y']].value_counts())\np0_A_counts = p0_A_counts.rename(columns={0:'count'}).reset_index()\n\np0_A_counts_grid = p0_A_counts.pivot(index='p0_y', columns='p0_x', values='count')\n# fill locations that are never occupied by p0 with 0.0001\n# to avoid divide by 0 error in the following lines\np0_A_counts_grid = p0_A_counts_grid.fillna(0.0001)\n\np0_A_grid_adj = p0_A_grid / p0_A_counts_grid\np0_A_grid_adj","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:11:05.091657Z","iopub.execute_input":"2022-10-04T08:11:05.092106Z","iopub.status.idle":"2022-10-04T08:11:06.266805Z","shell.execute_reply.started":"2022-10-04T08:11:05.092070Z","shell.execute_reply":"2022-10-04T08:11:06.265331Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8.3, 12))\nsns.heatmap(p0_A_grid_adj)","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:11:12.916640Z","iopub.execute_input":"2022-10-04T08:11:12.917801Z","iopub.status.idle":"2022-10-04T08:11:14.000207Z","shell.execute_reply.started":"2022-10-04T08:11:12.917761Z","shell.execute_reply":"2022-10-04T08:11:13.998332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below is the code wrapped in a function.","metadata":{}},{"cell_type":"code","source":"def xy_player_heatmap(df, player, scoring_team):\n    '''\n    df: DataFrame with columns for \"[players]_pos_x\", \"[players]_pos_y\",\n        and \"team_[A|B]_scoring_within_10sec\"\n    player: (str) a player (p0, p1, p2, p3, p4, p5, p6) to visualize\n    scoring_team: (str) \"A\" or \"B\"\n    '''\n    # [(-83, -81), (-81, -79), ..., (-3, -1), (-1, 1), ... (81, 83)]\n    x_intervals = [(2*x-3, 2*x-1) for x in range(-40, 43)]\n    x_bins = pd.IntervalIndex.from_tuples(x_intervals)\n    # [(-120, -118), (-118, -116) ... (-2, 0), (0, 2), ... (116, 118), (118, 120)]\n    y_intervals = [(2*y-2, 2*y) for y in range(-59, 61)] \n    y_bins = pd.IntervalIndex.from_tuples(y_intervals)\n    \n    df[f\"{player}_x\"] = pd.cut(df[f\"{player}_pos_x\"], bins=x_bins)\n    df[f\"{player}_y\"] = pd.cut(df[f\"{player}_pos_y\"], bins=y_bins)\n    \n    player_df = pd.DataFrame(\n                    df.groupby([f\"{player}_x\", f\"{player}_y\"])\n                    [f'team_{scoring_team}_scoring_within_10sec'].sum()\n                    ).reset_index()\n    player_grid = player_df.pivot(index=f\"{player}_y\",\n                                  columns=f\"{player}_x\",\n                                  values=f\"team_{scoring_team}_scoring_within_10sec\")\n    \n    counts = pd.DataFrame(df[[f\"{player}_x\", f\"{player}_y\"]].value_counts())\n    counts = counts.rename(columns={0:'count'}).reset_index()\n    counts_grid = counts.pivot(index=f\"{player}_y\", columns=f\"{player}_x\",\n                               values='count')\n    counts_grid = counts_grid.fillna(0.0001)\n    \n    player_grid_adj = player_grid/counts_grid\n    \n    fig, ax = plt.subplots(1, 2, figsize=(8.3*2, 12), sharey=True)\n    sns.heatmap(player_grid, ax=ax[0])\n    sns.heatmap(player_grid_adj, ax=ax[1])\n    ax[0].set_title(\"not adjusted for each location's frequency\")\n    ax[1].set_title(\"adjusted for each location's frequency\")\n    fig.suptitle(f\"Where {player} is within 10s of team {scoring_team} scoring\", fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:21:48.392541Z","iopub.execute_input":"2022-10-04T08:21:48.394254Z","iopub.status.idle":"2022-10-04T08:21:48.408824Z","shell.execute_reply.started":"2022-10-04T08:21:48.394201Z","shell.execute_reply":"2022-10-04T08:21:48.406594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xy_player_heatmap(df, 'p1', 'A')","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:22:25.789569Z","iopub.execute_input":"2022-10-04T08:22:25.789970Z","iopub.status.idle":"2022-10-04T08:23:13.797905Z","shell.execute_reply.started":"2022-10-04T08:22:25.789938Z","shell.execute_reply":"2022-10-04T08:23:13.796857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XY positions of attackers before goals\nUsing the method described above, here is a visualization of the positions of p0, p1, p2 within 10 seconds of team A scoring","metadata":{}},{"cell_type":"code","source":"def xy_players_heatmap(df, players, scoring_team):\n    '''\n    df: DataFrame with columns for \"[players]_pos_x\", \"[players]_pos_y\",\n        and \"team_[A|B]_scoring_within_10sec\"\n    players: (list of str) players (p0, p1, p2, p3, p4, p5, p6) to visualize\n    scoring_team: (str) \"A\" or \"B\"\n    '''\n    # [(-83, -81), (-81, -79), ..., (-3, -1), (-1, 1), ... (81, 83)]\n    x_intervals = [(2*x-3, 2*x-1) for x in range(-40, 43)]\n    x_bins = pd.IntervalIndex.from_tuples(x_intervals)\n    # [(-120, -118), (-118, -116) ... (-2, 0), (0, 2), ... (116, 118), (118, 120)]\n    y_intervals = [(2*y-2, 2*y) for y in range(-59, 61)] \n    y_bins = pd.IntervalIndex.from_tuples(y_intervals)\n    \n    player_grid_dict = {}\n    counts_grid_dict = {}\n    for player in players:\n        df[f\"{player}_x\"] = pd.cut(df[f\"{player}_pos_x\"], bins=x_bins)\n        df[f\"{player}_y\"] = pd.cut(df[f\"{player}_pos_y\"], bins=y_bins)\n        \n         \n        player_df = pd.DataFrame(\n                        df.groupby([f\"{player}_x\", f\"{player}_y\"])\n                        [f'team_{scoring_team}_scoring_within_10sec'].sum()\n                        ).reset_index()\n        player_grid_dict[f\"{player}\"] = player_df.pivot(index=f\"{player}_y\",\n                                                             columns=f\"{player}_x\",\n                                                             values=f\"team_{scoring_team}_scoring_within_10sec\")\n        \n    \n        counts = pd.DataFrame(df[[f\"{player}_x\", f\"{player}_y\"]].value_counts())\n        counts = counts.rename(columns={0:'count'}).reset_index()\n        counts_grid_dict[f\"{player}\"] = counts.pivot(index=f\"{player}_y\",\n                                                    columns=f\"{player}_x\",\n                                                    values='count').fillna(0.0001)\n        \n    \n    players_grid_sum = sum(player_grid_dict.values()) / len(player_grid_dict)\n    counts_grid_sum = sum(counts_grid_dict.values()) / len(counts_grid_dict)\n    \n    players_grid_adj = players_grid_sum/counts_grid_sum\n    \n    fig, ax = plt.subplots(1, 2, figsize=(8.3*2, 12), sharey=True)\n    sns.heatmap(players_grid_sum, ax=ax[0])\n    sns.heatmap(players_grid_adj, ax=ax[1])\n    ax[0].set_title(\"not adjusted for each location's frequency\")\n    ax[1].set_title(\"adjusted for each location's frequency\")\n    fig.suptitle(f\"Locations of {players} within 10s of team {scoring_team} scoring\", fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:27:20.383190Z","iopub.execute_input":"2022-10-04T08:27:20.383556Z","iopub.status.idle":"2022-10-04T08:27:20.397918Z","shell.execute_reply.started":"2022-10-04T08:27:20.383530Z","shell.execute_reply":"2022-10-04T08:27:20.397069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parquet_folder = '/kaggle/input/tps-oct-2022-compressed-parquet-files'\nd = {\"t0\":None,\"t1\":None,\"t2\":None,\"t3\":None,\"t4\":None,\n     \"t5\":None,\"t6\":None,\"t7\":None,\"t8\":None,\"t9\":None}\nfor i, k in enumerate(d):\n    d[k] = pd.read_parquet(os.path.join(parquet_folder,\n                                        f\"train_{i}.parquet.gzip\"),\n                           columns=['p0_pos_x', 'p0_pos_y',\n                                    'p1_pos_x', 'p1_pos_y',\n                                    'p2_pos_x', 'p2_pos_y',\n                                    'team_A_scoring_within_10sec'])\n\ndf = pd.concat([d['t0'], d['t1'], d['t2'], d['t3'], d['t4'],\n                    d['t5'], d['t6'], d['t7'], d['t8'], d['t9']]).reset_index()\n\nxy_players_heatmap(df, [\"p0\", \"p1\", \"p2\"], \"A\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:29:03.960465Z","iopub.execute_input":"2022-10-04T08:29:03.960817Z","iopub.status.idle":"2022-10-04T08:31:27.312920Z","shell.execute_reply.started":"2022-10-04T08:29:03.960792Z","shell.execute_reply":"2022-10-04T08:31:27.311178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Similarly, position of players p3, p4, p5 within 10 seconds of team B scoring:","metadata":{}},{"cell_type":"code","source":"parquet_folder = '/kaggle/input/tps-oct-2022-compressed-parquet-files'\nd = {\"t0\":None,\"t1\":None,\"t2\":None,\"t3\":None,\"t4\":None,\n     \"t5\":None,\"t6\":None,\"t7\":None,\"t8\":None,\"t9\":None}\nfor i, k in enumerate(d):\n    d[k] = pd.read_parquet(os.path.join(parquet_folder,\n                                        f\"train_{i}.parquet.gzip\"),\n                           columns=['p3_pos_x', 'p3_pos_y',\n                                    'p4_pos_x', 'p4_pos_y',\n                                    'p5_pos_x', 'p5_pos_y',\n                                    'team_B_scoring_within_10sec'])\n\ndf = pd.concat([d['t0'], d['t1'], d['t2'], d['t3'], d['t4'],\n                    d['t5'], d['t6'], d['t7'], d['t8'], d['t9']]).reset_index()\n\nxy_players_heatmap(df, [\"p3\", \"p4\", \"p5\"], \"B\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:34:01.136574Z","iopub.execute_input":"2022-10-04T08:34:01.137324Z","iopub.status.idle":"2022-10-04T08:36:19.380337Z","shell.execute_reply.started":"2022-10-04T08:34:01.137280Z","shell.execute_reply":"2022-10-04T08:36:19.378590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XY positions of defenders before goals","metadata":{}},{"cell_type":"code","source":"parquet_folder = '/kaggle/input/tps-oct-2022-compressed-parquet-files'\nd = {\"t0\":None,\"t1\":None,\"t2\":None,\"t3\":None,\"t4\":None,\n     \"t5\":None,\"t6\":None,\"t7\":None,\"t8\":None,\"t9\":None}\nfor i, k in enumerate(d):\n    d[k] = pd.read_parquet(os.path.join(parquet_folder,\n                                        f\"train_{i}.parquet.gzip\"),\n                           columns=['p3_pos_x', 'p3_pos_y',\n                                    'p4_pos_x', 'p4_pos_y',\n                                    'p5_pos_x', 'p5_pos_y',\n                                    'team_A_scoring_within_10sec'])\n\ndf = pd.concat([d['t0'], d['t1'], d['t2'], d['t3'], d['t4'],\n                    d['t5'], d['t6'], d['t7'], d['t8'], d['t9']]).reset_index()\n\nxy_players_heatmap(df, [\"p3\", \"p4\", \"p5\"], \"A\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:36:19.382480Z","iopub.execute_input":"2022-10-04T08:36:19.382779Z","iopub.status.idle":"2022-10-04T08:38:39.677262Z","shell.execute_reply.started":"2022-10-04T08:36:19.382753Z","shell.execute_reply":"2022-10-04T08:38:39.675911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parquet_folder = '/kaggle/input/tps-oct-2022-compressed-parquet-files'\nd = {\"t0\":None,\"t1\":None,\"t2\":None,\"t3\":None,\"t4\":None,\n     \"t5\":None,\"t6\":None,\"t7\":None,\"t8\":None,\"t9\":None}\nfor i, k in enumerate(d):\n    d[k] = pd.read_parquet(os.path.join(parquet_folder,\n                                        f\"train_{i}.parquet.gzip\"),\n                           columns=['p0_pos_x', 'p0_pos_y',\n                                    'p1_pos_x', 'p1_pos_y',\n                                    'p2_pos_x', 'p2_pos_y',\n                                    'team_B_scoring_within_10sec'])\n\ndf = pd.concat([d['t0'], d['t1'], d['t2'], d['t3'], d['t4'],\n                    d['t5'], d['t6'], d['t7'], d['t8'], d['t9']]).reset_index()\n\nxy_players_heatmap(df, [\"p0\", \"p1\", \"p2\"], \"B\")","metadata":{"execution":{"iopub.status.busy":"2022-10-04T08:40:12.657249Z","iopub.execute_input":"2022-10-04T08:40:12.657675Z","iopub.status.idle":"2022-10-04T08:42:35.131890Z","shell.execute_reply.started":"2022-10-04T08:40:12.657640Z","shell.execute_reply":"2022-10-04T08:42:35.130365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}