{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch as T\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-19T17:31:01.508292Z","iopub.execute_input":"2021-12-19T17:31:01.50878Z","iopub.status.idle":"2021-12-19T17:31:02.493142Z","shell.execute_reply.started":"2021-12-19T17:31:01.508683Z","shell.execute_reply":"2021-12-19T17:31:02.492337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ID to Names**","metadata":{}},{"cell_type":"code","source":"player_data = pd.read_csv('../input/nfl-big-data-bowl-2022/players.csv')\nnames = {player['nflId']: player['displayName'] for _, player in player_data.iterrows()}\nnames[-1.0] = 'Unknown'","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:40:18.108787Z","iopub.execute_input":"2021-12-19T17:40:18.10908Z","iopub.status.idle":"2021-12-19T17:40:18.268649Z","shell.execute_reply.started":"2021-12-19T17:40:18.10904Z","shell.execute_reply":"2021-12-19T17:40:18.267666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Field Goal Kickers**","metadata":{}},{"cell_type":"code","source":"plays = pd.read_csv('../input/nfl-big-data-bowl-2022/plays.csv')\nkickers = plays[['gameId','playId', 'possessionTeam', 'specialTeamsPlayType','specialTeamsResult','kickerId', 'yardlineSide', 'yardlineNumber']].copy()\nkickers = kickers.loc[(kickers['specialTeamsPlayType'] == 'Field Goal') | (kickers['specialTeamsPlayType'] == 'Extra Point')]\nkickers['kickerId'] = kickers['kickerId'].fillna(-1)\nkickers = kickers.sort_values(by=['kickerId'])\nprint(kickers)","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:40:22.412803Z","iopub.execute_input":"2021-12-19T17:40:22.413273Z","iopub.status.idle":"2021-12-19T17:40:22.596741Z","shell.execute_reply.started":"2021-12-19T17:40:22.413228Z","shell.execute_reply":"2021-12-19T17:40:22.595821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Kicker Analysis**","metadata":{}},{"cell_type":"code","source":"raw ={}\nprev = None\nfor _, data in kickers.iterrows():\n    kicker = names[data['kickerId']]\n    if kicker != prev:\n        if prev is not None: raw[kicker] = stats\n        stats = {'Field Goal':[0, 0, 0], 'Extra Point':[0, 0, 0]}\n        furthest = 0\n    prev = kicker\n    \n    kick_type = data['specialTeamsPlayType']\n    length = data['yardlineNumber']+17 if data['yardlineSide'] != data['possessionTeam'] else 117-data['yardlineNumber']\n    result = 0 if 'No' in data['specialTeamsResult'] else 1\n    \n    stats[kick_type][0] += 1\n    stats[kick_type][1] += result\n    stats[kick_type][2] += length\n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-11T08:06:38.56857Z","iopub.execute_input":"2021-12-11T08:06:38.569147Z","iopub.status.idle":"2021-12-11T08:06:38.959359Z","shell.execute_reply.started":"2021-12-11T08:06:38.56911Z","shell.execute_reply":"2021-12-11T08:06:38.958505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = {'kicker':[],\n         'fieldGoalAttempts':[],'fieldGoalsMade':[],'averageDistance':[], 'fgAccuracy':[], 'diffScore':[],\n        'extraPointAttempts':[], 'extraPointsMade':[],'expAccuracy':[]}\nfor k, stats in raw.items():\n    if stats['Extra Point'][0]+stats['Field Goal'][0] >= 30:\n        final['kicker'].append(k)\n        fg = stats['Field Goal']\n        exp = stats['Extra Point']\n        \n        final['fieldGoalAttempts'].append(fg[0])\n        final['fieldGoalsMade'].append(fg[1])\n        final['averageDistance'].append(fg[2]/fg[0])\n        final['fgAccuracy'].append(fg[1]/fg[0]*100)\n        final['diffScore'].append((fg[2]/fg[0])*(fg[1]/fg[0]))\n        \n        \n        final['extraPointAttempts'].append(exp[0])\n        final['extraPointsMade'].append(exp[1])\n        final['expAccuracy'].append(exp[1]/exp[0]*100)\n        \nkicker_stats = pd.DataFrame(final)\nprint(kicker_stats)\n        ","metadata":{"execution":{"iopub.status.busy":"2021-12-11T08:06:45.976071Z","iopub.execute_input":"2021-12-11T08:06:45.976559Z","iopub.status.idle":"2021-12-11T08:06:45.996373Z","shell.execute_reply.started":"2021-12-11T08:06:45.976508Z","shell.execute_reply":"2021-12-11T08:06:45.995293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_dist = kicker_stats['averageDistance'].idxmax()\naccurate = kicker_stats['fgAccuracy'].idxmax()\ngreedy_best = kicker_stats['diffScore'].idxmax()\nprint(kicker_stats[\"kicker\"][max_dist])\nprint(kicker_stats[\"kicker\"][accurate])\nprint(kicker_stats[\"kicker\"][greedy_best])","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:40:25.766641Z","iopub.execute_input":"2021-12-19T17:40:25.767573Z","iopub.status.idle":"2021-12-19T17:40:25.848478Z","shell.execute_reply.started":"2021-12-19T17:40:25.767525Z","shell.execute_reply":"2021-12-19T17:40:25.847362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Top Kickers**","metadata":{}},{"cell_type":"markdown","source":"**Average Field Gol Accuracy Per Yard Line**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**What are the hardest field goal situations?**\n\n- Score\n    - close game\n    - changes # of possessions\n    - kick gives lead to kicking team\n- Time remaining\n    - Near end of game\n    - Near end of half\n- Distance\n- Home/Away\n\nGeneral idea: We are trying to find feactures that make kicks difficult so we can use those features to give kicks a rating. The qualitative interperetation of these features should be something we believe makes a kick more difficult. Quantitatively, this means that the feature's value should near one when this feature is present on a given kick. The computer will then tell us how corrolated the given feature is to the outcome of the kick. When we yield coefficients, their closeness to one tells us the significance of its respecive feature. (Should also add features we don't think impact difficuty as a control)","metadata":{}},{"cell_type":"code","source":"plays = pd.read_csv('../input/nfl-big-data-bowl-2022/plays.csv')\n# plays = plays.loc[(plays['specialTeamsPlayType'] == 'Field Goal') | (plays['specialTeamsPlayType'] == 'Extra Point')]\nplays = plays.loc[(plays['specialTeamsPlayType'] == 'Field Goal') | (plays['specialTeamsPlayType']== 'Extra Point')]\nplays = plays.loc[(plays['specialTeamsResult'] == 'Kick Attempt Good') | (plays['specialTeamsResult'] == 'Kick Attempt No Good')]","metadata":{"execution":{"iopub.status.busy":"2021-12-08T19:34:39.957926Z","iopub.execute_input":"2021-12-08T19:34:39.958239Z","iopub.status.idle":"2021-12-08T19:34:40.082335Z","shell.execute_reply.started":"2021-12-08T19:34:39.958207Z","shell.execute_reply":"2021-12-08T19:34:40.08133Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"game_data = pd.read_csv('../input/nfl-big-data-bowl-2022/games.csv')\n\ndata = {'result':[], # 0 if miss 1 if made (labels)\n        # features\n        'closeness':[], # f(x)=(0.5)^((x/8)^2) Near one when one possession or less, near zero otherwise\n        'forLead':[], # 1 if kick gives a lead, 0 otherwise\n        'changeNumPossessions':[], # |scorediff+3| % 8 <=3\n        'time':[], #fraction of way through game (0 at start, 1 at conclusion)\n        'endHalf':[], # 0 if <10 sec in Q2, 0 otherwise\n        'distance':[], # distance / 76 (longest attempt in NFL history)\n        'home':[], # 0 if away, 1 if home\n       }\n\nexp = []\n\nfor i, row in plays.iterrows():\n    data['result'].append(float(row['specialTeamsResult'] == 'Kick Attempt Good'))\n    # find home team and away team\n    game_id = row['gameId']\n    home = game_data.loc[game_data['gameId'] == game_id,'homeTeamAbbr'].iloc[0] == row['possessionTeam']\n    data['home'].append(float(home))\n    \n    length = row['yardlineNumber']+17 if row['yardlineSide'] != row['possessionTeam'] else 117-row['yardlineNumber']\n    data['distance'].append(length/76)\n    \n    game_clock, quarter = row['gameClock'], row['quarter']\n    data['endHalf'].append(float(int(game_clock[3:5]) <= 30 and game_clock[:2] == '00' and quarter == 2))\n    # (time elapsed + minutes left + secondsleft/60) / 60 mins\n    data['time'].append( ((quarter-1)*15 + int(game_clock[:2]) + int(game_clock[3:5])/60) / 60)\n    \n    if home: \n        score, oppscore = row['preSnapHomeScore'], row['preSnapVisitorScore']\n    else:\n        score, oppscore = row['preSnapVisitorScore'], row['preSnapHomeScore']\n        \n    scorediff = score - oppscore\n    data['changeNumPossessions'].append(float(abs(scorediff+3)%8 <= 3))\n    data['forLead'].append(float(scorediff>=-2 and scorediff <=0))\n    data['closeness'].append(0.5**((scorediff/8)**2))\n    \n    exp.append(row['specialTeamsPlayType'] == 'Field Goal')\n    \n    \nraw_data = pd.DataFrame(data)\nprint(raw_data)","metadata":{"execution":{"iopub.status.busy":"2021-12-08T20:37:27.790466Z","iopub.execute_input":"2021-12-08T20:37:27.790925Z","iopub.status.idle":"2021-12-08T20:37:30.433757Z","shell.execute_reply.started":"2021-12-08T20:37:27.790894Z","shell.execute_reply":"2021-12-08T20:37:30.432487Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(0)\n# labels = points.iloc[:,:1]\n# fg_data = points.iloc[:,1:]\n\n# test_indices = np.random.choice(6054, 605)\n# test_labels = labels.loc[raw_data.index[test_indices]]\n# test_data = fg_data.loc[raw_data.index[test_indices]]\n\n# test_indices = set(test_indices)\n# train_indices = [i for i in range(6055) if i not in test_indices]\n# train_labels = labels.loc[raw_data.index[train_indices]]\n# train_data = fg_data.loc[raw_data.index[train_indices]]\n\nall_miss = raw_data.loc[raw_data['result'] == float(0)]\nall_make = raw_data.loc[raw_data['result'] == float(1)]\nsample_made = all_make.sample(585)\n\ndataset = pd.concat([all_miss, sample_made])\n\nlabels = dataset.iloc[:,:1]\nfg_data = dataset.iloc[:,1:]\n\ntest_indices = np.random.choice(1169, 234)\ntest_labels = labels.loc[dataset.index[test_indices]]\ntest_data = fg_data.loc[dataset.index[test_indices]]\n\ntest_indices = set(test_indices)\ntrain_indices = [i for i in range(1170) if i not in test_indices]\ntrain_labels = labels.loc[dataset.index[train_indices]]\ntrain_data = fg_data.loc[dataset.index[train_indices]]","metadata":{"execution":{"iopub.status.busy":"2021-10-16T01:49:43.006171Z","iopub.execute_input":"2021-10-16T01:49:43.006422Z","iopub.status.idle":"2021-10-16T01:49:43.025231Z","shell.execute_reply.started":"2021-10-16T01:49:43.006394Z","shell.execute_reply":"2021-10-16T01:49:43.024117Z"},"_kg_hide-input":false,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = T.device('cpu')\n\nclass Dataset(T.utils.data.Dataset):\n\n    def __init__(self, data, labels):\n\n        self.x_data = T.tensor(data.to_numpy(), dtype=T.float32).to(device)\n        self.y_data = T.tensor(labels.to_numpy(), dtype=T.float32).to(device)\n        self.y_data = self.y_data.reshape(len(self.y_data),1)\n\n    def __len__(self):\n        return len(self.x_data)\n\n    def __getitem__(self, idx):\n        if T.is_tensor(idx):\n            idx = idx.tolist()\n        preds = self.x_data[idx,:]  # idx rows, all 4 cols\n        lbl = self.y_data[idx,:]    # idx rows, the 1 col\n        sample = { 'predictors' : preds, 'target' : lbl }\n        return sample\n\n    \nclass Net(T.nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.hid1 = T.nn.Linear(7, 4)  # 7-(4-4)-1\n        self.hid2 = T.nn.Linear(4, 4)\n        self.oupt = T.nn.Linear(4, 1)\n\n        T.nn.init.xavier_uniform_(self.hid1.weight)\n        T.nn.init.zeros_(self.hid1.bias)\n        T.nn.init.xavier_uniform_(self.hid2.weight)\n        T.nn.init.zeros_(self.hid2.bias)\n        T.nn.init.xavier_uniform_(self.oupt.weight)\n        T.nn.init.zeros_(self.oupt.bias)\n\n    def forward(self, x):\n        z = T.tanh(self.hid1(x)) \n        z = T.tanh(self.hid2(z))\n        z = T.sigmoid(self.oupt(z))\n        return z\n\n\ndef acc_coarse(model, ds):\n    inpts = ds[:]['predictors']  # all rows\n    targets = ds[:]['target']    # all target 0s and 1s\n    with T.no_grad():\n        oupts = model(inpts)       # all computed ouputs\n    pred_y = oupts >= 0.5        # tensor of 0s and 1s\n    num_correct = T.sum(targets==pred_y)  # tensor\n    acc = (num_correct.item() * 1.0 / len(ds))  # scalar\n    return acc","metadata":{"execution":{"iopub.status.busy":"2021-10-16T01:49:43.027671Z","iopub.execute_input":"2021-10-16T01:49:43.028043Z","iopub.status.idle":"2021-10-16T01:49:43.050049Z","shell.execute_reply.started":"2021-10-16T01:49:43.02801Z","shell.execute_reply":"2021-10-16T01:49:43.049264Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Without bootstrapping","metadata":{}},{"cell_type":"code","source":"T.manual_seed(0)\ntrain_ds = Dataset(train_data, train_labels)  # all rows\n\nbat_size = 10\ntrain_ldr = T.utils.data.DataLoader(train_ds, batch_size=bat_size, shuffle=True)\n\nnet = Net().to(device)\nnet = net.train()  # set training mode\n\nlrn_rate = 0.01\nloss_obj = T.nn.BCELoss()  # binary cross entropy\noptimizer = T.optim.SGD(net.parameters(),\n  lr=lrn_rate)\n\nfor epoch in range(0, 100):\n    epoch_loss = 0.0  # sum of avg loss/item/batch\n\n    for (batch_idx, batch) in enumerate(train_ldr):\n        X = batch['predictors']  # [10,4]  inputs\n        Y = batch['target']      # [10,1]  targets\n        optimizer.zero_grad()\n        oupt = net(X)            # [10,1]  computeds \n\n        loss_val = loss_obj(oupt, Y)   # a tensor\n        epoch_loss += loss_val.item()  # accumulate\n        loss_val.backward()  # compute all gradients\n        optimizer.step()     # update all wts, biases\n\n    if epoch % 10 == 0:  \n        print(\"epoch = %4d   loss = %0.4f\" % (epoch, epoch_loss))\n","metadata":{"execution":{"iopub.status.busy":"2021-10-16T01:49:43.051483Z","iopub.execute_input":"2021-10-16T01:49:43.051919Z","iopub.status.idle":"2021-10-16T01:49:51.87749Z","shell.execute_reply.started":"2021-10-16T01:49:43.051811Z","shell.execute_reply":"2021-10-16T01:49:51.876491Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net = net.eval()\nacc = acc_coarse(net, Dataset(test_data, test_labels))\nprint(\"\\nAccuracy = %0.4f\" % acc)\n\nprint(\"Saving trained model state dict \")\npath = \"banknote_sd_model.pth\"\nT.save(net.state_dict(), path)\n\nprint(sum(test_labels.to_numpy())/test_labels.shape[0])\n","metadata":{"execution":{"iopub.status.busy":"2021-10-16T02:12:54.326428Z","iopub.execute_input":"2021-10-16T02:12:54.326787Z","iopub.status.idle":"2021-10-16T02:12:54.33845Z","shell.execute_reply.started":"2021-10-16T02:12:54.32675Z","shell.execute_reply":"2021-10-16T02:12:54.337287Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Interperet\n\n- Tells us that not all kicks of x distance are of equal difficulty\n- create function(difficulty score) = % chance\n- interperet in terms of EPA\n- With more data, (weather, crowd noise, etc.) can get even better judgement\n- Better judge kickers (who is most clutch) (define threshhold for 'clutch' kick, ","metadata":{}},{"cell_type":"code","source":"desired_set = all_make\ntot = 0\nfor _ in range(10):\n    sample = desired_set.sample(100)\n    labs = sample.iloc[:,:1]\n    rand_data = sample.iloc[:,1:]\n\n    net = net.eval()\n    acc = acc_coarse(net, Dataset(rand_data, labs))\n    print(\"\\nAccuracy = %0.4f\" % acc)\n    tot += acc\nprint()\nprint(tot/10)","metadata":{"execution":{"iopub.status.busy":"2021-10-16T02:11:36.98649Z","iopub.execute_input":"2021-10-16T02:11:36.986858Z","iopub.status.idle":"2021-10-16T02:11:37.013488Z","shell.execute_reply.started":"2021-10-16T02:11:36.986821Z","shell.execute_reply":"2021-10-16T02:11:37.012405Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Try training with bootstrapping and weighted labels","metadata":{}},{"cell_type":"code","source":"def get_new_sample():\n    sample = raw_data.sample(int(raw_data.shape[0]*.2))\n    labels = sample.iloc[:,:1]\n    data = sample.iloc[:,1:]\n\n    return data, labels\n\ndef test_avg_accuracy(net):\n    tot = 0\n    for _ in range(10):\n        rand_data, labs = get_new_sample()\n\n        net = net.eval()\n        acc = acc_coarse(net, Dataset(rand_data, labs))\n        print(\"\\nAccuracy = %0.4f\" % acc)\n        tot += acc\n    print()\n    print(tot/10)","metadata":{"execution":{"iopub.status.busy":"2021-10-16T02:47:03.503622Z","iopub.execute_input":"2021-10-16T02:47:03.503966Z","iopub.status.idle":"2021-10-16T02:47:03.512222Z","shell.execute_reply.started":"2021-10-16T02:47:03.503934Z","shell.execute_reply":"2021-10-16T02:47:03.511168Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"T.manual_seed(0)\n\nbias = sum(raw_data['result'].to_numpy())/len(raw_data['result'])\n\nnet2 = Net().to(device)\nnet2 = net2.train()  # set training mode\n\nlrn_rate = 0.01\n#loss_obj = T.nn.BCELoss()  # binary cross entropy\noptimizer = T.optim.SGD(net2.parameters(),\n  lr=lrn_rate)\nfor _ in range(10):\n    train_data, train_labels = get_new_sample()\n    train_ds = Dataset(train_data, train_labels)  # all rows\n\n    bat_size = 10\n    train_ldr = T.utils.data.DataLoader(train_ds, batch_size=bat_size, shuffle=True)\n    for epoch in range(0, 100):\n        epoch_loss = 0.0  # sum of avg loss/item/batch\n\n        for (batch_idx, batch) in enumerate(train_ldr):\n            X = batch['predictors']  # [10,4]  inputs\n            Y = batch['target']      # [10,1]  targets\n\n            optimizer.zero_grad()\n            oupt = net2(X)            # [10,1]  computeds \n            \n            weights = [[bias] if target.item() == 0.0 else [1-bias] for target in Y]\n            class_weights = T.FloatTensor(weights)\n#             print(jumba)\n            loss_obj = T.nn.BCELoss(weight=class_weights)\n\n            \n            loss_val = loss_obj(oupt, Y)   # a tensor\n            epoch_loss += loss_val.item()  # accumulate\n            loss_val.backward()  # compute all gradients\n            optimizer.step()     # update all wts, biases\n\n        if epoch % 99 == 0:  \n            print(\"epoch = %4d   loss = %0.4f\" % (epoch, epoch_loss))\n            \ntest_avg_accuracy(net2)","metadata":{"execution":{"iopub.status.busy":"2021-10-16T03:10:06.680238Z","iopub.execute_input":"2021-10-16T03:10:06.680741Z","iopub.status.idle":"2021-10-16T03:12:14.588634Z","shell.execute_reply.started":"2021-10-16T03:10:06.680706Z","shell.execute_reply":"2021-10-16T03:12:14.588006Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"desired_set = all_miss\ntot = 0\nfor _ in range(10):\n    sample = desired_set.sample(100)\n    labs = sample.iloc[:,:1]\n    rand_data = sample.iloc[:,1:]\n\n    net = net.eval()\n    acc = acc_coarse(net2, Dataset(rand_data, labs))\n    print(\"\\nAccuracy = %0.4f\" % acc)\n    tot += acc\nprint()\nprint(tot/10)","metadata":{"execution":{"iopub.status.busy":"2021-10-16T03:13:31.378701Z","iopub.execute_input":"2021-10-16T03:13:31.379075Z","iopub.status.idle":"2021-10-16T03:13:31.403729Z","shell.execute_reply.started":"2021-10-16T03:13:31.379032Z","shell.execute_reply":"2021-10-16T03:13:31.402427Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next idea:\nmodel a scoring function\n- PCA to visualize\n- Overlay kicker salary proj onto x-y (highlight corrolation)\n    - displays possible application","metadata":{}},{"cell_type":"code","source":"def standardize_data(arr):\n           \n    rows, columns = arr.shape\n    \n    standardizedArray = np.zeros(shape=(rows, columns))\n    tempArray = np.zeros(rows)\n    \n    for column in range(columns):\n        \n        mean = np.mean(X[:,column])\n        std = np.std(X[:,column])\n        tempArray = np.empty(0)\n        \n        for element in X[:,column]:\n            \n            tempArray = np.append(tempArray, ((element - mean) / std))\n \n        standardizedArray[:,column] = tempArray\n    \n    return standardizedArray","metadata":{"execution":{"iopub.status.busy":"2021-12-08T19:37:20.180364Z","iopub.execute_input":"2021-12-08T19:37:20.180638Z","iopub.status.idle":"2021-12-08T19:37:20.187941Z","shell.execute_reply.started":"2021-12-08T19:37:20.180611Z","shell.execute_reply":"2021-12-08T19:37:20.187017Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = raw_data.iloc[:, 1:].values\ny = raw_data.result.values\nX = standardize_data(X)\n\ncovariance_matrix = np.cov(X.T)\n\neigen_values, eigen_vectors = np.linalg.eig(covariance_matrix)\nprint(\"Eigenvector: \\n\",eigen_vectors,\"\\n\")\nprint(\"Eigenvalues: \\n\", eigen_values, \"\\n\")\n\n\neigen_vectors = eigen_vectors[np.argsort(eigen_values)[::-1]]\neigen_values = np.sort(eigen_values)[::-1]\n\nvariance_explained = []\nfor i in eigen_values:\n     variance_explained.append((i/sum(eigen_values))*100)\nprint(variance_explained)\nprint()\n\ncumulative_variance_explained = np.cumsum(variance_explained)\nprint(cumulative_variance_explained)\nprint()\n\n\nn = 2 # number of eigenvectors kept\nprojection_matrix = (eigen_vectors.T[:][:2]).T\nprint(projection_matrix)\n\nX_pca = X.dot(projection_matrix)\nprint(X_pca)","metadata":{"execution":{"iopub.status.busy":"2021-12-08T19:37:22.737854Z","iopub.execute_input":"2021-12-08T19:37:22.738171Z","iopub.status.idle":"2021-12-08T19:37:23.270214Z","shell.execute_reply.started":"2021-12-08T19:37:22.738137Z","shell.execute_reply":"2021-12-08T19:37:23.269275Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# colors = ['#0000FF' if l else '#FF0000' for l in exp]\nfor key in data:\n#     if key == 'time': continue\n    print(key)\n    colors = ['#' + 3*hex(int(255//2*(dist)))[2:] for dist in data[key]]\n    plt.figure()\n    plt.scatter(X_pca[:,0],X_pca[:,1], c=colors)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:43:04.802572Z","iopub.execute_input":"2021-12-19T17:43:04.803108Z","iopub.status.idle":"2021-12-19T17:43:04.818089Z","shell.execute_reply.started":"2021-12-19T17:43:04.803074Z","shell.execute_reply":"2021-12-19T17:43:04.817335Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PCA Conclusion: outside of distance, the features chosen don't influence the outcome of the kicks","metadata":{}},{"cell_type":"markdown","source":"Final idea: ranking that control for stadium played in.\n\nNotoriously harder to kick in places like Buffalo, Pittsburgh, etc. Lets make ranking that adjusts for these, since kickers of these teams kick more kicks in the harder places to kick.\n\n1. Establish that some stadiums are harder to kick in than others\n\n2. Find accuracy in each stadium for each kicker and average accuracy for all kickers in each stadium\n\n3. Compute accuracy - avg for each kicker in each stadium. Averaging this will give more 'fair' ranking of kickers","metadata":{}},{"cell_type":"code","source":"# Map eatch team abbreviation string to a unique index between 0-31\ntidx = {\n    'ARI': 0,\n    'BUF': 1,\n    'CAR': 2,\n    'CLE': 3,\n    'DAL': 4,\n    'GB': 5,\n    'HOU': 6,\n    'KC': 7,\n    'LAC': 8,\n    'MIN': 9,\n    'NE': 10,\n    'NYJ': 11,\n    'LV': 12,\n    'SF': 13,\n    'SEA': 14,\n    'WAS': 15,\n    'ATL': 16,\n    'BAL': 17,\n    'CHI': 18,\n    'CIN': 19,\n    'DEN': 20,\n    'DET': 21,\n    'IND': 22,\n    'JAX': 23,\n    'LAR': 24,\n    'MIA': 25,\n    'NO': 26,\n    'NYG': 27,\n    'PHI': 28,\n    'PIT': 29,\n    'TB': 30,\n    'TEN': 31\n}\nteam_abbr = [\n    'ARI',\n    'BUF',\n    'CAR',\n    'CLE',\n    'DAL',\n    'GB',\n    'HOU',\n    'KC',\n    'LAC',\n    'MIN',\n    'NE',\n    'NYJ',\n    'LV',\n    'SF',\n    'SEA',\n    'WAS',\n    'ATL',\n    'BAL',\n    'CHI',\n    'CIN',\n    'DEN',\n    'DET',\n    'IND',\n    'JAX',\n    'LAR',\n    'MIA',\n    'NO',\n    'NYG',\n    'PHI',\n    'PIT',\n    'TB',\n    'TEN'\n]","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:41:15.995255Z","iopub.execute_input":"2021-12-19T17:41:15.995549Z","iopub.status.idle":"2021-12-19T17:41:16.002844Z","shell.execute_reply.started":"2021-12-19T17:41:15.99552Z","shell.execute_reply":"2021-12-19T17:41:16.001994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = pd.read_csv('../input/nfl-big-data-bowl-2022/games.csv')\n\nkicker_make = {}\nkicker_att = {}\nmade_per_stadium = {}\natt_per_stadium = {}\nfor i, kick in kickers.iterrows():\n    k = kick['kickerId']\n    if names[k] == 'Unknown': continue\n    home = g.loc[g['gameId'] == kick['gameId']]['homeTeamAbbr'].values[0]\n    if home == 'OAK': home = 'LV'\n    if home == 'LA': home = 'LAR'\n    att_per_stadium[home] = att_per_stadium.get(home,0) + 1\n    attempts = kicker_att.setdefault(k,[0 for _ in range(32)])\n    attempts[tidx[home]] += 1\n    \n    made = kicker_make.setdefault(k,[0 for _ in range(32)])\n    if 'No' not in kick['specialTeamsResult']: \n        made_per_stadium[home] = made_per_stadium.get(home,0) + 1\n        made[tidx[home]] += 1\n        \n    \n    \nacc_per_stadium = {}\nfor home in made_per_stadium:\n    acc_per_stadium[home] = made_per_stadium[home]/att_per_stadium[home]\n    \nkicker_per_stadium = {}\nfor kicker in kicker_make:\n    att = kicker_att[kicker]\n    made = kicker_make[kicker]\n    kicker_per_stadium[names[kicker]] = [made[i]/att[i] if att[i] != 0 else 0 for i in range(32)]\n    \nkick_acc = {kicker: sum(vals)/32 for kicker, vals in kicker_per_stadium.items()}\nacc = pd.DataFrame.from_dict(kick_acc, orient='index', columns = ['accuracy']).sort_values(by='accuracy')\n    \nacc_per_stadium = pd.DataFrame.from_dict(acc_per_stadium, orient='index', columns=['accuracy']).sort_values(by='accuracy')\nprint(acc_per_stadium)\n\nkicker_per_stadium = pd.DataFrame.from_dict(kicker_per_stadium, orient='index', columns=team_abbr)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:41:34.085413Z","iopub.execute_input":"2021-12-19T17:41:34.085709Z","iopub.status.idle":"2021-12-19T17:41:36.929425Z","shell.execute_reply.started":"2021-12-19T17:41:34.085675Z","shell.execute_reply":"2021-12-19T17:41:36.928504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hex colors for each team\nteam_to_color = {\n    'ARI': '#97233F',\n    'BUF': '#00338D',\n    'CAR': '#0085CA',\n    'CLE': '#311D00',\n    'DAL': '#041E42',\n    'GB': '#203731',\n    'HOU': '#03202F',\n    'KC': '#E31837',\n    'LAC': '#0080C6',\n    'MIN': '#4F2683',\n    'NE': '#002244',\n    'NYJ': '#125740',\n    'LV': '#000000',\n    'SF': '#B3995D',\n    'SEA': '#69BE28',\n    'WAS': '#773141',\n    'ATL': '#A71930',\n    'BAL': '#241773',\n    'CHI': '#C83803',\n    'CIN': '#FB4F14',\n    'DEN': '#002244',\n    'DET': '#0076B6',\n    'IND': '#002C5F',\n    'JAX': '#006778',\n    'LAR': '#003594',\n    'MIA': '#008E97',\n    'NO': '#D3BC8D',\n    'NYG': '#0B2265',\n    'PHI': '#004C54',\n    'PIT': '#FFB612',\n    'TB': '#D50A0A',\n    'TEN': '#4B92DB'\n}","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:44:05.980932Z","iopub.execute_input":"2021-12-19T17:44:05.981374Z","iopub.status.idle":"2021-12-19T17:44:05.987779Z","shell.execute_reply.started":"2021-12-19T17:44:05.981337Z","shell.execute_reply":"2021-12-19T17:44:05.987224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = [team_to_color[name] for name in acc_per_stadium.index]\nplt.figure(figsize=(16,4))\nplt.scatter(acc_per_stadium.index,acc_per_stadium, c=colors)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:44:10.530147Z","iopub.execute_input":"2021-12-19T17:44:10.530561Z","iopub.status.idle":"2021-12-19T17:44:10.82685Z","shell.execute_reply.started":"2021-12-19T17:44:10.530532Z","shell.execute_reply":"2021-12-19T17:44:10.826062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that Washington has the hardest stadium to kick in, while Houston has the easiest stadium to kick in.","metadata":{}},{"cell_type":"code","source":"print(kicker_per_stadium)","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:44:18.166091Z","iopub.execute_input":"2021-12-19T17:44:18.166501Z","iopub.status.idle":"2021-12-19T17:44:18.191501Z","shell.execute_reply.started":"2021-12-19T17:44:18.166472Z","shell.execute_reply":"2021-12-19T17:44:18.190674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, this matrix is rather sparse. This is because the dataset only considers two years of kicks, and not all kickers kick in each stadium within two years. To combat this, we can find a low rank approximation of this matrix to get an estimate of kickers accuracies in other stadiums.","metadata":{}},{"cell_type":"code","source":"u, s, vt = np.linalg.svd(kicker_per_stadium)\nrank = 6\nu_t, s_t, vt_t = u[:,:rank], np.diag(s[:rank]), vt[:rank,:]\nlra = np.matmul(np.matmul(u_t, s_t), vt_t)\n# lra = np.where(lra<0, 0, lra)\nkicker_approx = pd.DataFrame(lra, columns=team_abbr, index=kicker_per_stadium.index)\n\navg_acc = [0 for _ in range(32)]\nfor team in team_abbr:\n    avg_acc[tidx[team]] = sum(kicker_approx[team].values)/61\n\nest_acc = pd.DataFrame(avg_acc, index=team_abbr, columns=['accuracy']).sort_values(by='accuracy')\n# print(est_acc)\n\n# colors = [team_to_color[name] for name in est_acc.index]\n# plt.figure(figsize=(16,4))\n# plt.scatter(est_acc.index,est_acc, c=colors)\n# plt.show()\n\nres =  []\nfor _, row in kicker_approx.iterrows():\n    vals = row.values\n    diff = [vals[tidx[team]]-est_acc.loc[team].values[0] for team in est_acc.index]\n    avg = [sum(diff)/32]\n    res.append(diff+avg)\n\nfinal = pd.DataFrame(res, columns=list(est_acc.index) + ['avgAcc'], index=kicker_approx.index).sort_values(by='avgAcc')\n\n# print(final['avgAcc'])\n\n# plt.figure(figsize=(16,4))\n# plt.scatter(final.index, final['avgAcc'])\n# plt.show()\n\n\n\nfinal_idx = list(final.index)\nacc_idx = list(acc.index)\n\nmovements = []\nfor kicker in final.index:\n    move = final_idx.index(kicker)-acc_idx.index(kicker)\n    print(kicker, move)\n    movements.append((kicker,move))\n\nprint('\\n', 'Biggest Mover: ', max(movements, key=lambda x: x[1]))\nprint('Avg movement: ', sum([abs(move[1]) for move in movements])/len(movements), '\\n')\n","metadata":{"execution":{"iopub.status.busy":"2021-12-19T17:55:43.099966Z","iopub.execute_input":"2021-12-19T17:55:43.100291Z","iopub.status.idle":"2021-12-19T17:55:43.261051Z","shell.execute_reply.started":"2021-12-19T17:55:43.100261Z","shell.execute_reply":"2021-12-19T17:55:43.260071Z"},"trusted":true},"execution_count":null,"outputs":[]}]}