{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Abstract\n\nThis notebook aims to suggest a method to split fold with less player duplication.\nIf the train/test dataset are chosen from completely different data sources, your CV doesn't show the correct generalization performance of your model. For example, if the players/teams in the train and test dataset differs, your CV might only shows the overfitted performance of your model to the specific players/teams in the train dataset.\n\nThis notebook compares similarity of fold pairs of suggested method to that of usual StratifiedGroupKFold. The result shows the suggested method reduced the average simiarity ~4.6% (24.0% -> 19.4%).","metadata":{}},{"cell_type":"code","source":"!pip install scikit-learn-extra nb-black","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-09T09:36:32.697662Z","iopub.execute_input":"2023-03-09T09:36:32.698379Z","iopub.status.idle":"2023-03-09T09:36:48.785911Z","shell.execute_reply.started":"2023-03-09T09:36:32.698339Z","shell.execute_reply":"2023-03-09T09:36:48.784438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%load_ext lab_black\n%load_ext autoreload\n%autoreload 2\n\nfrom itertools import product\n\nimport matplotlib.pyplot as plt\nimport polars as pl\nimport numpy as np\nimport seaborn as sns\nfrom sklearn_extra.cluster import KMedoids\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import StratifiedGroupKFold","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-09T09:36:48.788617Z","iopub.execute_input":"2023-03-09T09:36:48.788993Z","iopub.status.idle":"2023-03-09T09:36:50.140075Z","shell.execute_reply.started":"2023-03-09T09:36:48.788955Z","shell.execute_reply":"2023-03-09T09:36:50.139014Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_base = pl.read_csv(\n    \"../input/nfl-player-contact-detection/train_labels.csv\", null_values=[\"G\"]\n).with_columns(\n    [\n        pl.col(\"game_play\").str.split(\"_\").arr.get(0).cast(pl.Int32).alias(\"game_key\"),\n        pl.col(\"game_play\").str.split(\"_\").arr.get(1).cast(pl.Int32).alias(\"play_id\"),\n    ]\n)\ntr_base_swapped = tr_base.filter(\n    pl.col(\"nfl_player_id_2\").is_null().is_not()\n).with_columns(\n    [\n        pl.col(\"nfl_player_id_1\").alias(\"nfl_player_id_2\"),\n        pl.col(\"nfl_player_id_2\").alias(\"nfl_player_id_1\"),\n    ]\n)\ntr_base_ext = pl.concat([tr_base, tr_base_swapped])\ncontact_count = tr_base_ext.groupby(\"game_key\").agg(pl.col(\"contact\").sum())\ndisplay(contact_count.head())\n\ngame_feats = (\n    tr_base_ext.with_columns(pl.lit(1).cast(pl.UInt8).alias(\"player\"))\n    .pivot(\n        index=\"game_key\",\n        columns=\"nfl_player_id_1\",\n        values=\"player\",\n        aggregate_function=\"max\",\n    )\n    .fill_null(0)\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-09T09:36:50.142211Z","iopub.execute_input":"2023-03-09T09:36:50.142684Z","iopub.status.idle":"2023-03-09T09:36:57.181170Z","shell.execute_reply.started":"2023-03-09T09:36:50.142636Z","shell.execute_reply":"2023-03-09T09:36:57.179926Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Method\n\nThe suggested method is as follows:\n\n1. make one-hot vector of players per `game_key`\n2. reduce dimension of the feature by PCA(n_components=32)\n3. aplly KMedoids clustering (n_clusters=5)\n\nI think PCA part is optional. I manually chosen the random seed of KMedoids clustering so that the number of contact events per each folds becomes roughly even.","metadata":{}},{"cell_type":"markdown","source":"## Experiment\n\nSimilarity is calculated by this equation:\n\n$$\\text{sim}(i, j) = \\frac{\\#\\left(\\text{players}(i) \\cap \\text{players}(j)\\right)}{\\#\\left(\\text{players}(i) \\cup \\text{players}(j)\\right)} $$\n\n### Similarity over fold pairs if using StratifiedGroupKFold","metadata":{}},{"cell_type":"code","source":"X = tr_base_ext.select(\"step\")\ny = tr_base_ext[\"contact\"]\ngroups = tr_base_ext[\"game_key\"]\nsgkf = StratifiedGroupKFold(n_splits=5)\n\n\nfold_split_sgkf = np.empty(len(tr_base_ext), dtype=np.uint8)\nfor fold, (train_idxs, valid_idxs) in enumerate(sgkf.split(X, y, groups)):\n    fold_split_sgkf[valid_idxs] = fold\n\nfold_split_sgkf = tr_base_ext.select(\n    [pl.col(\"game_key\"), pl.Series(fold_split_sgkf).alias(\"fold\")]\n).unique()\ndisplay(\n    fold_split_sgkf.join(contact_count, on=\"game_key\", how=\"left\")\n    .groupby(\"fold\")\n    .agg(pl.col(\"contact\").sum())\n    .sort(\"fold\")\n)\n\nfold_vec = (\n    fold_split_sgkf.join(game_feats, on=\"game_key\")\n    .drop(\"game_key\")\n    .groupby(\"fold\")\n    .agg(pl.all().max())\n    .sort(\"fold\")\n    .drop(\"fold\")\n    .to_numpy()\n)\nsim_ij_skgf = np.zeros((5, 5))\nfor i, j in product(range(5), range(5)):\n    sim_ij_skgf[i, j] = (fold_vec[i] & fold_vec[j]).sum() / (\n        fold_vec[i] | fold_vec[j]\n    ).sum()\n\n\n_, ax = plt.subplots()\nsns.heatmap(sim_ij_skgf, ax=ax, annot=True, vmin=0, vmax=1)\nax.set(xlabel=\"fold\", ylabel=\"fold\", title=\"#Duplicated Players per each folds\")\nplt.show()\nfor i in range(5):\n    sim_ij_skgf[i, i] = np.nan\nprint(f\"Average similarity: {np.nanmean(sim_ij_skgf):.5f}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:37:33.084522Z","iopub.execute_input":"2023-03-09T09:37:33.085389Z","iopub.status.idle":"2023-03-09T09:37:49.949100Z","shell.execute_reply.started":"2023-03-09T09:37:33.085341Z","shell.execute_reply":"2023-03-09T09:37:49.948164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Similarity over fold pairs if game_key is clustered by its players","metadata":{}},{"cell_type":"code","source":"X = game_feats.select(game_feats.columns[1:])\ny = game_feats.select(game_feats.columns[0])\n\npca = PCA(n_components=32, random_state=38749)\nX_reduced = pca.fit_transform(X.transpose())\n\nk_medoids_pca = KMedoids(n_clusters=5, random_state=28).fit(X_reduced)\n\nfold_split = game_feats.select(\n    [pl.col(\"game_key\"), pl.Series(k_medoids_pca.labels_).cast(pl.UInt8).alias(\"fold\")]\n)\ndisplay(\n    fold_split.join(contact_count, on=\"game_key\", how=\"left\")\n    .groupby(\"fold\")\n    .agg(pl.col(\"contact\").sum())\n    .sort(\"fold\")\n)\nfold_vec = (\n    fold_split.join(game_feats, on=\"game_key\")\n    .drop(\"game_key\")\n    .groupby(\"fold\")\n    .agg(pl.all().max())\n    .sort(\"fold\")\n    .drop(\"fold\")\n    .to_numpy()\n)\nsim_ij = np.zeros((5, 5))\nfor i, j in product(range(5), range(5)):\n    sim_ij[i, j] = (fold_vec[i] & fold_vec[j]).sum() / (fold_vec[i] | fold_vec[j]).sum()\n\n\n_, ax = plt.subplots()\nsns.heatmap(sim_ij, ax=ax, annot=True, vmin=0, vmax=1)\nax.set(\n    xlabel=\"fold\",\n    ylabel=\"fold\",\n    title=\"#Duplicated Players per each folds\",\n)\nplt.show()\nfor i in range(5):\n    sim_ij[i, i] = np.nan\nprint(f\"Average similarity: {np.nanmean(sim_ij):.5f}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:37:57.050340Z","iopub.execute_input":"2023-03-09T09:37:57.051457Z","iopub.status.idle":"2023-03-09T09:37:57.540725Z","shell.execute_reply.started":"2023-03-09T09:37:57.051360Z","shell.execute_reply":"2023-03-09T09:37:57.539468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusion\n\nIn this notebook, I suggested a simple fold split method that considers duplication of players in each games.\nThe experiment result shows the suggested method decrease duplication of players per each fold about 4.6%.\n\n## Future Works\n\nI left the following for future works:\n\n* more sofilsticated way to control the number of contact labels for each folds\n* more direct method to reduce duplication of players per each games","metadata":{}}]}