{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# K - Nearest Neighbors\n\nThe idea behind this approach was simple, that there are simply too many possibilities a game can move towards after any given instant screenshot of that game. What else do we have simply too much of? Training data. \n\nK-Nearest Neighbors is perhaps the simplest model that scores closest to 0.2~, a good score. For any given instant that we need to predict for, we find the closest K frames in the training data, and average how many of those ended up scoring a goal. It is difficult to estimate a good value for K, as estimations made on training subsamples don't seem to work well when the model is run on the entire dataset.\n\nOne approach to try is, with a large value of K, try to strech the 'predicted' probabilities. This is because, the larger the value of K, more the predicted probabilities shift closer to 0 overall. We could try slightly boosting few largest probabilities to reduce the overall logloss error.","metadata":{}},{"cell_type":"code","source":"!pip install functorch","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-27T12:12:35.861569Z","iopub.execute_input":"2022-10-27T12:12:35.862529Z","iopub.status.idle":"2022-10-27T12:13:55.655253Z","shell.execute_reply.started":"2022-10-27T12:12:35.862422Z","shell.execute_reply":"2022-10-27T12:13:55.654062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport functorch\nfrom tqdm import tqdm, trange\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-27T12:13:55.658570Z","iopub.execute_input":"2022-10-27T12:13:55.658957Z","iopub.status.idle":"2022-10-27T12:13:56.569450Z","shell.execute_reply.started":"2022-10-27T12:13:55.658915Z","shell.execute_reply":"2022-10-27T12:13:56.568542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\n\ndef get_train_data():\n    df = []\n    for i in trange(10):\n        df.append(pd.read_feather(f'../input/sorting-players-positions/train_{i}_sorted.ftr'))\n    df = pd.concat(df)\n    if DEBUG:\n        return df.sample(frac=0.01)\n    return df.sample(frac=1)\n\ndef get_test_data():\n    df =  pd.read_feather('../input/sorting-players-positions/test_sorted.ftr')\n    \n    if DEBUG:\n        return df.sample(frac=0.1)\n    \n    return df\n\ntest = get_test_data().fillna(0)\n\nFEATURES = list(test.columns)\nFEATURES.remove('id')\n\ntrain = get_train_data()\ntrainY = train[['team_A_scoring_within_10sec', 'team_B_scoring_within_10sec']]\n\ntrain = train[FEATURES].fillna(0)\ntest = test[FEATURES]","metadata":{"execution":{"iopub.status.busy":"2022-10-27T12:13:56.570772Z","iopub.execute_input":"2022-10-27T12:13:56.571653Z","iopub.status.idle":"2022-10-27T12:15:10.859165Z","shell.execute_reply.started":"2022-10-27T12:13:56.571615Z","shell.execute_reply":"2022-10-27T12:15:10.858165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class KNN():\n    def __init__(self, K=3, check_Ks=False):\n        self.K = K\n        self.check_Ks = check_Ks\n    \n    def fit(self, X, Y):\n        self.X = X\n        self.Y = Y\n        \n    def predict(self, tX):\n        dists = torch.sqrt(torch.sum((self.X - tX)**2, axis=1))\n        \n        min_indices = torch.topk(dists, self.K, largest=False, sorted=False).indices\n        del dists\n        \n        probaA = self.Y[min_indices, 0].sum() / len(min_indices)\n        probaB = self.Y[min_indices, 1].sum() / len(min_indices)\n        del min_indices\n        \n        return [probaA, probaB]\n    \n    \ndef save(predsA, predsB):\n    sub = pd.DataFrame({\n    'team_A_scoring_within_10sec': [preds.cpu().item() for preds in predsA],\n    'team_B_scoring_within_10sec': [preds.cpu().item() for preds in predsB]\n    })\n    sub['id'] = sub.index.values\n\n    sub = sub.iloc[:, [2, 0, 1]]\n    sub.to_csv('submission.csv', index=False)\n        \n","metadata":{"execution":{"iopub.status.busy":"2022-10-27T12:15:10.860575Z","iopub.execute_input":"2022-10-27T12:15:10.861220Z","iopub.status.idle":"2022-10-27T12:15:10.873196Z","shell.execute_reply.started":"2022-10-27T12:15:10.861183Z","shell.execute_reply":"2022-10-27T12:15:10.872287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cpu')\nif torch.cuda.is_available():\n    device = torch.device('cuda')\n\n\ntraintensor = torch.tensor(train.values).to(torch.float32).to(device)\ntrainYtensor = torch.tensor(trainY.values).to(torch.float32).to(device)\ntestensor = torch.tensor(test.values).to(torch.float32).to(device)\n\ndel train, trainY, test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-27T12:15:10.876132Z","iopub.execute_input":"2022-10-27T12:15:10.876698Z","iopub.status.idle":"2022-10-27T12:15:21.990367Z","shell.execute_reply.started":"2022-10-27T12:15:10.876661Z","shell.execute_reply":"2022-10-27T12:15:21.989293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nmodel = KNN(2501)\nmodel.fit(traintensor, trainYtensor)\n\npredsB = []\n\nrows = testensor.shape[0]\n\npredsA = []\npredsB = []\n\ntestloader = torch.utils.data.DataLoader(testensor, batch_size=1000, shuffle=False)\n\nvpredict = functorch.vmap(model.predict)\ndone_so_far = 0\n\nfor batch in testloader:    \n    if done_so_far and done_so_far % 5_000 == 0:\n        save(predsA, predsB)\n    \n    gc.collect()\n    for row in batch:\n        temp_preds = model.predict(row)\n        predsA.append(temp_preds[0])\n        predsB.append(temp_preds[1])\n        \n    done_so_far += 1000\n    print(f'Done with {done_so_far} Rows')\n    \n\n    ","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-27T12:15:21.991668Z","iopub.execute_input":"2022-10-27T12:15:21.993506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del traintensor, trainYtensor, testensor\ngc.collect()\n\nsub = pd.DataFrame({\n    'team_A_scoring_within_10sec': [preds.cpu().item() for preds in predsA],\n    'team_B_scoring_within_10sec': [preds.cpu().item() for preds in predsB]\n})\nsub['id'] = sub.index.values\n\nsub = sub.iloc[:, [2, 0, 1]]\nsub.head()\n\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(sub.head())\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}