{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Objective\n\nIn this notebook, I'll extract 10% of the game plays to use them for training neural networks.","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pylab as plt\n\nfrom sklearn.metrics import matthews_corrcoef\n\nSEED = 19951204\nCREATE_FRAMES_DF = True\nEXTRACT_FRAMES = True","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:28:07.880162Z","iopub.execute_input":"2022-12-21T10:28:07.881026Z","iopub.status.idle":"2022-12-21T10:28:08.595547Z","shell.execute_reply.started":"2022-12-21T10:28:07.880923Z","shell.execute_reply":"2022-12-21T10:28:08.594357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Files","metadata":{}},{"cell_type":"code","source":"# Read in data files\nBASE_DIR = \"../input/nfl-player-contact-detection\"\n\n# Labels and sample submission\nlabels = pd.read_csv(f\"{BASE_DIR}/train_labels.csv\", parse_dates=[\"datetime\"])\n\nss = pd.read_csv(f\"{BASE_DIR}/sample_submission.csv\")\n\n# Player tracking data\ntr_tracking = pd.read_csv(\n    f\"{BASE_DIR}/train_player_tracking.csv\", parse_dates=[\"datetime\"]\n)\nte_tracking = pd.read_csv(\n    f\"{BASE_DIR}/test_player_tracking.csv\", parse_dates=[\"datetime\"]\n)\n\n# Baseline helmet detection labels\ntr_helmets = pd.read_csv(f\"{BASE_DIR}/train_baseline_helmets.csv\")\nte_helmets = pd.read_csv(f\"{BASE_DIR}/test_baseline_helmets.csv\")\n\n# Video metadata with start/stop timestamps\ntr_video_metadata = pd.read_csv(\n    \"../input/nfl-player-contact-detection/train_video_metadata.csv\",\n    parse_dates=[\"start_time\", \"end_time\", \"snap_time\"],\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:28:08.597784Z","iopub.execute_input":"2022-12-21T10:28:08.598135Z","iopub.status.idle":"2022-12-21T10:28:55.473642Z","shell.execute_reply.started":"2022-12-21T10:28:08.598103Z","shell.execute_reply":"2022-12-21T10:28:55.472455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Source: https://www.kaggle.com/code/robikscube/nfl-player-contact-detection-getting-started\n\ndef compute_distance(df, tr_tracking, merge_col=\"datetime\"):\n    \"\"\"\n    Merges tracking data on player1 and 2 and computes the distance.\n    \"\"\"\n    df_combo = (\n        df.astype({\"nfl_player_id_1\": \"str\"})\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\", \"x_position\", \"y_position\"]\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_1\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .rename(columns={\"x_position\": \"x_position_1\", \"y_position\": \"y_position_1\"})\n        .drop(\"nfl_player_id\", axis=1)\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\", \"x_position\", \"y_position\"]\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_2\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .drop(\"nfl_player_id\", axis=1)\n        .rename(columns={\"x_position\": \"x_position_2\", \"y_position\": \"y_position_2\"})\n        .copy()\n    )\n\n    df_combo[\"distance\"] = np.sqrt(\n        np.square(df_combo[\"x_position_1\"] - df_combo[\"x_position_2\"])\n        + np.square(df_combo[\"y_position_1\"] - df_combo[\"y_position_2\"])\n    )\n    return df_combo\n\n\ndef add_contact_id(df):\n    # Create contact ids\n    df[\"contact_id\"] = (\n        df[\"game_play\"]\n        + \"_\"\n        + df[\"step\"].astype(\"str\")\n        + \"_\"\n        + df[\"nfl_player_id_1\"].astype(\"str\")\n        + \"_\"\n        + df[\"nfl_player_id_2\"].astype(\"str\")\n    )\n    return df\n\n\ndef expand_contact_id(df):\n    \"\"\"\n    Splits out contact_id into seperate columns.\n    \"\"\"\n    df[\"game_play\"] = df[\"contact_id\"].str[:12]\n    df[\"step\"] = df[\"contact_id\"].str.split(\"_\").str[-3].astype(\"int\")\n    df[\"nfl_player_id_1\"] = df[\"contact_id\"].str.split(\"_\").str[-2]\n    df[\"nfl_player_id_2\"] = df[\"contact_id\"].str.split(\"_\").str[-1]\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:28:55.475097Z","iopub.execute_input":"2022-12-21T10:28:55.475442Z","iopub.status.idle":"2022-12-21T10:28:55.490463Z","shell.execute_reply.started":"2022-12-21T10:28:55.475410Z","shell.execute_reply":"2022-12-21T10:28:55.489113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_combo = compute_distance(labels, tr_tracking)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:28:55.493469Z","iopub.execute_input":"2022-12-21T10:28:55.493879Z","iopub.status.idle":"2022-12-21T10:29:11.776621Z","shell.execute_reply.started":"2022-12-21T10:28:55.493844Z","shell.execute_reply":"2022-12-21T10:29:11.775771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold\n\nnp.random.seed(SEED)\n\nkf = GroupKFold()\nkf_dict = {}\n\nfor i, (train_index, test_index) in enumerate(kf.split(tr_video_metadata, None, tr_video_metadata['game_key'])):\n    print(f\"Fold {i}:\")\n    kf_dict[i] = {'train_games': list(tr_video_metadata.iloc[train_index].game_key.unique()),\n                  'val_games': list(tr_video_metadata.iloc[test_index].game_key.unique())}","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:29:11.778042Z","iopub.execute_input":"2022-12-21T10:29:11.778931Z","iopub.status.idle":"2022-12-21T10:29:11.806177Z","shell.execute_reply.started":"2022-12-21T10:29:11.778885Z","shell.execute_reply":"2022-12-21T10:29:11.804904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I should save the validation data games in order to use them further on during validation in different strategies.","metadata":{}},{"cell_type":"code","source":"import pickle\n\nwith open('kf_dict', 'wb') as f:\n    pickle.dump(kf_dict, f)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:29:11.807667Z","iopub.execute_input":"2022-12-21T10:29:11.808047Z","iopub.status.idle":"2022-12-21T10:29:11.816464Z","shell.execute_reply.started":"2022-12-21T10:29:11.808013Z","shell.execute_reply":"2022-12-21T10:29:11.815498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('kf_dict', 'rb') as f:\n    kf_dict = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:29:11.818151Z","iopub.execute_input":"2022-12-21T10:29:11.818616Z","iopub.status.idle":"2022-12-21T10:29:11.828485Z","shell.execute_reply.started":"2022-12-21T10:29:11.818574Z","shell.execute_reply":"2022-12-21T10:29:11.827575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's extract the validation set as a starter. I'll extract from only sideline view for starters.","metadata":{}},{"cell_type":"code","source":"import subprocess, os\nfrom tqdm.notebook import tqdm\n\nval_games = kf_dict[0]['val_games']\n\nval_game_plays = tr_video_metadata.query('game_key in @val_games').game_play\n\nif EXTRACT_FRAMES:\n    !mkdir -p validation \n    !chmod 777 validation\n\n    for g in tqdm(val_game_plays):\n        g_paths = !ls /kaggle/input/nfl-player-contact-detection/train/$g*\n        for g_path in g_paths:\n            game_play = g_path.split('/')[-1].split('/')[-1][:-4]\n            !mkdir -p validation/$game_play && chmod 777 validation/$game_play\n            !ffmpeg -i \"$g_path\" -q:v 2 -f image2 \"validation/$game_play/frame-%04d.jpg\" -hide_banner -loglevel error","metadata":{"execution":{"iopub.status.busy":"2022-12-21T10:29:11.830189Z","iopub.execute_input":"2022-12-21T10:29:11.830563Z","iopub.status.idle":"2022-12-21T11:11:12.637202Z","shell.execute_reply.started":"2022-12-21T10:29:11.830531Z","shell.execute_reply":"2022-12-21T11:11:12.635517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's extract 10% of train game_plays","metadata":{}},{"cell_type":"code","source":"train_games = kf_dict[0]['train_games']\n\ntrain_game_plays = tr_video_metadata.query('game_key in @train_games').sample(frac=0.1, replace=False, random_state=SEED).game_play","metadata":{"execution":{"iopub.status.busy":"2022-12-21T11:11:12.639888Z","iopub.execute_input":"2022-12-21T11:11:12.640314Z","iopub.status.idle":"2022-12-21T11:11:12.656886Z","shell.execute_reply.started":"2022-12-21T11:11:12.640272Z","shell.execute_reply":"2022-12-21T11:11:12.655479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTRACT_FRAMES:\n    !mkdir -p train\n    !chmod 777 train\n\n    for g in train_game_plays:\n        g_paths = !ls /kaggle/input/nfl-player-contact-detection/train/$g*\n        for g_path in g_paths:\n            game_play = g_path.split('/')[-1].split('/')[-1][:-4]\n            !mkdir -p train/$game_play && chmod 777 train/$game_play\n            !ffmpeg -i \"$g_path\" -q:v 2 -f image2 \"train/$game_play/frame-%04d.jpg\" -hide_banner -loglevel error","metadata":{"execution":{"iopub.status.busy":"2022-12-21T11:11:12.660469Z","iopub.execute_input":"2022-12-21T11:11:12.660975Z","iopub.status.idle":"2022-12-21T11:16:09.585936Z","shell.execute_reply.started":"2022-12-21T11:11:12.660938Z","shell.execute_reply":"2022-12-21T11:16:09.584460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create frame dataframe\nframes_df = []\n\ngame_plays = list(train_game_plays) + list(val_game_plays)\n\n# Loop over game_plays\nfor game_play in tqdm(game_plays):\n\n    game_play_combo = df_combo.query('game_play == @game_play')\n    game_play_helmets = tr_helmets.query('game_play == @game_play')\n\n    # Loop over rows\n    for row in game_play_combo.query('nfl_player_id_2 != \"G\" and distance < 2').itertuples(index=False):\n        players = [int(row.nfl_player_id_1), int(row.nfl_player_id_2)]\n        f_min = 290 + row.step*6\n        f_max = 290 + row.step*6 + 6\n\n        for f in range(f_min, f_max):\n            \n            for view in ['Sideline', 'Endzone']:\n\n                # Extract players' helmet bboxes\n                crop_df = game_play_helmets.query('nfl_player_id in @players and \\\n                                                   frame == @f and \\\n                                                   view == @view')\n                if len(crop_df) > 0:\n\n                    # Calculate the center of the players bboxes\n                    top_left = np.array([crop_df['top'].min(), crop_df['left'].min()])\n\n                    bottom_right = np.array([crop_df['top'].max() + crop_df.loc[crop_df['top'].idxmax(), 'height'],\n                                             crop_df['left'].max() + crop_df.loc[crop_df['left'].idxmax(), 'width']])\n\n                    two_players_center = (top_left + bottom_right) // 2\n\n                    # Save the frame data \n                    frame_row = dict(game_play=row.game_play, \n                                     nfl_player_id_1=players[0], \n                                     nfl_player_id_2=players[1],\n                                     frame=f, \n                                     step=row.step,\n                                     centery=two_players_center[0],\n                                     centerx=two_players_center[1],\n                                     distance=row.distance, \n                                     contact=row.contact)\n\n                    frames_df.append(frame_row)\n\n                else:\n                    break\n\nframes_df = pd.DataFrame(frames_df)\nframes_df = frames_df.dropna(subset=['distance'])","metadata":{"execution":{"iopub.status.busy":"2022-12-21T11:16:09.588047Z","iopub.execute_input":"2022-12-21T11:16:09.588451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frames_df.to_csv('frames_df.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}