{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import socket\n\nis_kaggle = _dh == ['/kaggle/working']\n\nif _dh != ['/kaggle/working']:\n    from my_utils import get_notebook_path\n    NB=get_notebook_path().split('/')[-1].split('.')[0]\nelse:\n    NB = ''\nDESCRIPTION='test notebook'\nHOST = socket.gethostname()\n\n\n\n# 変更箇所\n## 1.以下のif文内に埋め込む\n## 2.yoloの読み込みディレクトリ\n## 3.yoloのパラメーター\n## 4.yoloのファイル存在チェック(dataloaderから呼ぶやつ)\n## 5.YOLOのファイルパス注意（detect.py のﾋｷｽｳ）\nif is_kaggle:\n    HOST = '5aee93363caa'\n    NB = 'exp0135_cnn_multitask_0126base' #実行後にファイル命名を変えた関係でbugfixでなくなっている\n    \nNB, HOST","metadata":{"papermill":{"duration":0.112207,"end_time":"2021-09-27T02:54:59.505937","exception":false,"start_time":"2021-09-27T02:54:59.393730","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:21:07.555482Z","iopub.execute_input":"2023-02-16T15:21:07.556024Z","iopub.status.idle":"2023-02-16T15:21:07.601037Z","shell.execute_reply.started":"2023-02-16T15:21:07.555920Z","shell.execute_reply":"2023-02-16T15:21:07.600077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"papermill":{"duration":0.021702,"end_time":"2021-09-27T02:54:59.551604","exception":false,"start_time":"2021-09-27T02:54:59.529902","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:21:07.605701Z","iopub.execute_input":"2023-02-16T15:21:07.608073Z","iopub.status.idle":"2023-02-16T15:21:08.815950Z","shell.execute_reply.started":"2023-02-16T15:21:07.608034Z","shell.execute_reply":"2023-02-16T15:21:08.814591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if _dh != ['/kaggle/working']:\n    import mlflow\n    from logzero import logger\nelse:\n    import sys\n    sys.path.append('/kaggle/input/timm-pytorch-image-models/pytorch-image-models-master')\n\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport glob\nfrom sklearn.model_selection import StratifiedGroupKFold, StratifiedKFold, KFold\nfrom sklearn.metrics import roc_auc_score, accuracy_score, mean_squared_error, matthews_corrcoef\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom collections import OrderedDict\nimport random\nimport os\nimport gc\nimport shutil\nimport torch.optim as optim\nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nimport albumentations as albu\nimport pickle\nimport warnings\nwarnings.simplefilter('ignore')\n\nfrom sklearn.preprocessing import StandardScaler\nimport optuna\nimport seaborn as sns\n\ntqdm.pandas()\n\n\n    \nimport timm\n\nfrom pandarallel import pandarallel\nif _dh != ['/kaggle/working']:\n    pandarallel.initialize(nb_workers=30, use_memory_fs=False, progress_bar=False)\nelse:\n    pandarallel.initialize(nb_workers=4, use_memory_fs=False, progress_bar=True)\n    \n\nROOT_DIR = Path('../')\nif _dh != ['/kaggle/working']:\n    DATA_DIR = ROOT_DIR / Path('data')\nelse:\n    DATA_DIR = Path('/kaggle/input/nfl-player-contact-detection')\nOUTPUT_DIR = ROOT_DIR / 'output'\nCP_DIR = OUTPUT_DIR / 'checkpoint'\n\ndef to_pickle(filename, obj):\n    with open(filename, mode='wb') as f:\n        pickle.dump(obj, f)\n        \ndef unpickle(filename):\n    with open(filename, mode='rb') as fo:\n        p = pickle.load(fo)\n    return p ","metadata":{"papermill":{"duration":6.891314,"end_time":"2021-09-27T02:55:06.464626","exception":false,"start_time":"2021-09-27T02:54:59.573312","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:21:08.821439Z","iopub.execute_input":"2023-02-16T15:21:08.822161Z","iopub.status.idle":"2023-02-16T15:21:14.720360Z","shell.execute_reply.started":"2023-02-16T15:21:08.822115Z","shell.execute_reply":"2023-02-16T15:21:14.719237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\nprint(multiprocessing.cpu_count())\n\nclass Config:\n    N_LABEL = 1\n    N_FOLD = 3\n    RANDOM_SATE = 42\n    LR = 1.0e-05\n    MAX_LR = 1.0e-4\n    PATIENCE = 15\n    EPOCH = 12\n    BATCH_SIZE = 96\n    SKIP_EVALUATE_NUM = 0\n    BACK_BONE = 'tf_efficientnet_b0_ns'\n    RUN_FOLD_COUNT = 10\n    IMG_SIZE=128\n    T_MAX=20\n    ETA_MIN=3.0e-7\n    SCHEDULER_GAMMA=1.0\n    ACCUMULATION_STEMP=2\n    NUM_WORKERS=multiprocessing.cpu_count()\n    YOLO_SIZE='yolov5x.pt'\n    \n    \n\"\"\"class Config:\n    N_LABEL = 1\n    N_FOLD = 2\n    RANDOM_SATE = 42\n    LR = 1.0e-05\n    MAX_LR = 8.0e-5\n    PATIENCE = 15\n    EPOCH = 2\n    BATCH_SIZE = 1024\n    SKIP_EVALUATE_NUM = 0\n    BACK_BONE = 'tf_efficientnet_b0_ns'\n    RUN_FOLD_COUNT = 10\n    IMG_SIZE=16\n    T_MAX=20\n    ETA_MIN=3.0e-7\n    SCHEDULER_GAMMA=1.0\n    ACCUMULATION_STEMP=2\n    NUM_WORKERS=multiprocessing.cpu_count()\n    YOLO_SIZE='yolov5s.pt'\"\"\"\n    \n\ndef seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nseed_everything(seed=Config.RANDOM_SATE)","metadata":{"papermill":{"duration":0.038449,"end_time":"2021-09-27T02:55:06.632933","exception":false,"start_time":"2021-09-27T02:55:06.594484","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:21:14.723293Z","iopub.execute_input":"2023-02-16T15:21:14.723953Z","iopub.status.idle":"2023-02-16T15:21:14.738111Z","shell.execute_reply.started":"2023-02-16T15:21:14.723913Z","shell.execute_reply":"2023-02-16T15:21:14.736685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data読み込み","metadata":{}},{"cell_type":"code","source":"if not is_kaggle:\n    train_labels_df = pd.read_csv(DATA_DIR / Path('train_labels.csv'))\n    train_player_tracking_df = pd.read_csv(DATA_DIR / Path('train_player_tracking.csv'))\n    test_player_tracking_df = pd.read_csv(DATA_DIR / Path('test_player_tracking.csv'))\n    train_df = train_player_tracking_df[train_player_tracking_df['step'] >= -5].reset_index(drop=True)\n    test_df = test_player_tracking_df[test_player_tracking_df['step'] >= -5].reset_index(drop=True)\n\n    test_helmets = pd.read_csv(DATA_DIR / \"test_baseline_helmets.csv\")\n    test_video_metadata = pd.read_csv(DATA_DIR / \"test_video_metadata.csv\")\n\n    train_helmets = pd.read_csv(DATA_DIR / \"train_baseline_helmets.csv\")\n    train_video_metadata = pd.read_csv(DATA_DIR / \"train_video_metadata.csv\")\n\n    sample_submission_df = pd.read_csv(DATA_DIR / \"sample_submission.csv\")\n\n    display(train_df.head(2))\n    display(test_df.head(2))\n    display(train_labels_df.head(2))\n\n    del train_player_tracking_df, test_player_tracking_df\nelse:\n    train_player_tracking_df = pd.read_csv(DATA_DIR / Path('train_player_tracking.csv'))\n    test_player_tracking_df = pd.read_csv(DATA_DIR / Path('test_player_tracking.csv'))\n    train_df = train_player_tracking_df[train_player_tracking_df['step'] >= -5].reset_index(drop=True)\n    test_df = test_player_tracking_df[test_player_tracking_df['step'] >= -5].reset_index(drop=True)\n\n    test_helmets = pd.read_csv(DATA_DIR / \"test_baseline_helmets.csv\")\n    test_video_metadata = pd.read_csv(DATA_DIR / \"test_video_metadata.csv\")\n\n    sample_submission_df = pd.read_csv(DATA_DIR / \"sample_submission.csv\")\n\n    display(train_df.head(2))\n    display(test_df.head(2))\n\n    del train_player_tracking_df, test_player_tracking_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:21:14.740114Z","iopub.execute_input":"2023-02-16T15:21:14.740810Z","iopub.status.idle":"2023-02-16T15:21:19.010923Z","shell.execute_reply.started":"2023-02-16T15:21:14.740773Z","shell.execute_reply":"2023-02-16T15:21:19.009891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels_df = sample_submission_df.copy()\ntest_labels_df['nfl_player_id_2'] = test_labels_df['contact_id'].apply(lambda x : x.split('_')[4])\ntest_labels_df['step'] = test_labels_df['contact_id'].apply(lambda x : x.split('_')[2]).astype(int)\ntest_labels_df['game_play'] = test_labels_df['contact_id'].apply(lambda x : x.split('_')[0] + '_' + x.split('_')[1])\ntest_labels_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:21:19.012639Z","iopub.execute_input":"2023-02-16T15:21:19.013061Z","iopub.status.idle":"2023-02-16T15:21:19.124135Z","shell.execute_reply.started":"2023-02-16T15:21:19.013021Z","shell.execute_reply":"2023-02-16T15:21:19.123246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## video to image","metadata":{}},{"cell_type":"code","source":"if is_kaggle:\n    !rm -r ../work/frames\n    !mkdir -p ../work/frames\n\n    for video in tqdm(test_helmets.video.unique()):\n        if 'Endzone2' not in video:\n            !ffmpeg -i {DATA_DIR.resolve()}/test/{video} -q:v 2 -f image2 ../work/frames/{video}_%04d.jpg -hide_banner -loglevel error\n        \n\"\"\"for video in tqdm(train_helmets.video.unique()):\n    if 'Endzone2' not in video:\n        !ffmpeg -i {DATA_DIR.resolve()}/train/{video} -q:v 2 -f image2 ../work/frames/{video}_%04d.jpg -hide_banner -loglevel error\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:21:19.125588Z","iopub.execute_input":"2023-02-16T15:21:19.125905Z","iopub.status.idle":"2023-02-16T15:22:06.700783Z","shell.execute_reply.started":"2023-02-16T15:21:19.125878Z","shell.execute_reply":"2023-02-16T15:22:06.699515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### YOLOで選手検出\napiでちょくで呼び出すので必要なし","metadata":{}},{"cell_type":"code","source":"\"\"\"if is_kaggle:\n    !mkdir -p /kaggle/temp/yolo\n    %cd /kaggle/temp/yolo\n    !cp -r /kaggle/input/yolov5main ./ # DataSetによって切り替える\n    %cd ./yolov5main                   # DataSetによって切り替える\n    !mkdir runs\n    for video in tqdm(test_helmets['video'].unique(), smoothing=0):\n        if 'Endzone2' not in video:\n            !python detect.py --source /kaggle/input/nfl-player-contact-detection/test/{video} --weights yolov5s.pt --save-txt --iou-thres 0.5 --conf-thres=0.3 --classes 0 --name {video} --nosave\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.702301Z","iopub.execute_input":"2023-02-16T15:22:06.702671Z","iopub.status.idle":"2023-02-16T15:22:06.716471Z","shell.execute_reply.started":"2023-02-16T15:22:06.702635Z","shell.execute_reply":"2023-02-16T15:22:06.715253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"if is_kaggle:\n    %cd /kaggle/working\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.718005Z","iopub.execute_input":"2023-02-16T15:22:06.718289Z","iopub.status.idle":"2023-02-16T15:22:06.840838Z","shell.execute_reply.started":"2023-02-16T15:22:06.718259Z","shell.execute_reply":"2023-02-16T15:22:06.839615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    max_speed_dict = train_df.groupby('nfl_player_id')['speed'].max().to_dict()\n    max_acceleration_dict = train_df.groupby('nfl_player_id')['acceleration'].max().to_dict()\n    max_sa_dict = train_df.groupby('nfl_player_id')['acceleration'].max().to_dict()\n\n    train_df['player_max_speed'] = train_df['nfl_player_id'].map(max_speed_dict)\n    train_df['player_max_acceleration'] = train_df['nfl_player_id'].map(max_acceleration_dict)\n    train_df['player_max_sa'] = train_df['nfl_player_id'].map(max_sa_dict)\n\n    test_df['player_max_speed'] = test_df['nfl_player_id'].map(max_speed_dict)\n    test_df['player_max_acceleration'] = test_df['nfl_player_id'].map(max_acceleration_dict)\n    test_df['player_max_sa'] = test_df['nfl_player_id'].map(max_sa_dict)\nelse:\n    max_speed_dict = test_df.groupby('nfl_player_id')['speed'].max().to_dict()\n    max_acceleration_dict = test_df.groupby('nfl_player_id')['acceleration'].max().to_dict()\n    max_sa_dict = test_df.groupby('nfl_player_id')['acceleration'].max().to_dict()\n\n    train_max_speed_dict = train_df.groupby('nfl_player_id')['speed'].max().to_dict()\n    train_max_acceleration_dict = train_df.groupby('nfl_player_id')['acceleration'].max().to_dict()\n    train_max_sa_dict = train_df.groupby('nfl_player_id')['acceleration'].max().to_dict()\n\n    max_speed_dict.update(train_max_speed_dict)\n    max_acceleration_dict.update(train_max_acceleration_dict)\n    max_sa_dict.update(train_max_sa_dict)\n\n    test_df['player_max_speed'] = test_df['nfl_player_id'].map(max_speed_dict)\n    test_df['player_max_acceleration'] = test_df['nfl_player_id'].map(max_acceleration_dict)\n    test_df['player_max_sa'] = test_df['nfl_player_id'].map(max_sa_dict)\n    \n    del train_df\n","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.847085Z","iopub.execute_input":"2023-02-16T15:22:06.847477Z","iopub.status.idle":"2023-02-16T15:22:06.921807Z","shell.execute_reply.started":"2023-02-16T15:22:06.847430Z","shell.execute_reply":"2023-02-16T15:22:06.920804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## テーブル特徴量","metadata":{}},{"cell_type":"code","source":"def calc_centroid(df):\n    \n    df = df.merge(df.groupby(['game_play', 'step'])['x_position'].mean().to_frame().add_suffix('_centroid'), how='left', on=['game_play', 'step'])\n    df = df.merge(df.groupby(['game_play', 'step'])['y_position'].mean().to_frame().add_suffix('_centroid'), how='left', on=['game_play', 'step'])\n    \n    df['x_position_std'] = df['x_position'] - df['x_position_centroid']\n    df['y_position_std'] = df['y_position'] - df['y_position_centroid']\n    #重心からの距離\n    df['player_distance_centroid'] = np.sqrt((df['x_position'] - df['x_position_centroid'])**2 + (df['y_position'] - df['y_position_centroid'])**2)\n    \n    df = df.merge(df.groupby(['game_play', 'step', 'team'])['x_position'].mean().to_frame().add_suffix('_team_centroid'), how='left', on=['game_play', 'step', 'team'])\n    df = df.merge(df.groupby(['game_play', 'step', 'team'])['y_position'].mean().to_frame().add_suffix('_team_centroid'), how='left', on=['game_play', 'step', 'team'])\n    \n    df['x_position_team_std'] = df['x_position'] - df['x_position_team_centroid']\n    df['y_position_team_std'] = df['y_position'] - df['y_position_team_centroid']\n    #重心からの距離\n    df['player_distance_team_centroid'] = np.sqrt((df['x_position'] - df['x_position_team_centroid'])**2 + (df['y_position'] - df['y_position_team_centroid'])**2)\n    \n    return df\n\ndef calc_direction(df):\n    df['y_direction'] = df['direction'].apply(np.sin)\n    df['x_direction'] = df['direction'].apply(np.cos)\n    df['y_orientation'] = df['orientation'].apply(np.sin)\n    df['x_orientation'] = df['orientation'].apply(np.cos)\n    return df\n\ndef concat_history(df):\n    df = pd.concat([df,\n           df.groupby(['game_play', 'nfl_player_id'])[['x_position', 'y_position', 'distance', 'direction', 'orientation', 'acceleration', 'sa']].shift(1).add_suffix('_shift_1'),\n           df.groupby(['game_play', 'nfl_player_id'])[['x_position', 'y_position', 'distance', 'direction', 'orientation', 'acceleration', 'sa']].shift(-1).add_suffix('_shift_-1'),\n          ], axis=1)\n    \n    for shift in range(-5, 6):\n        df[f'speed_shift{shift}'] = df.groupby(['game_play', 'nfl_player_id'])['speed'].shift(shift)\n    \n    return df\n\ndef calc_bbox_size(df, helmets):\n    df['frame'] = (df['step']/10*59.94+5*59.94).astype('int')+1\n    \n    _helmets = helmets[helmets['view'] == 'Endzone'].copy()\n    _helmets['bbox_size_endzone'] = _helmets['width'] * _helmets['height']\n    mean_bbox_df = _helmets.groupby(['game_play', 'frame'])['bbox_size_endzone'].mean().to_frame('mean_bbox_size_endzone').reset_index()\n    df = df.merge(_helmets[['game_play', 'frame', 'nfl_player_id', 'bbox_size_endzone']], how='left', on=['game_play', 'frame', 'nfl_player_id'])\n    df = df.merge(mean_bbox_df, how='left', on=['game_play', 'frame'])\n    \n    _helmets = helmets[helmets['view'] == 'Sideline'].copy()\n    _helmets['bbox_size_sideline'] = _helmets['width'] * _helmets['height']\n    mean_bbox_df = _helmets.groupby(['game_play', 'frame'])['bbox_size_sideline'].mean().to_frame('mean_bbox_size_sideline').reset_index()\n    df = df.merge(_helmets[['game_play', 'frame', 'nfl_player_id', 'bbox_size_sideline']], how='left', on=['game_play', 'frame', 'nfl_player_id'])\n    df = df.merge(mean_bbox_df, how='left', on=['game_play', 'frame'])\n    \n    \n    df = df.drop('frame', axis=1)\n    \n    return df\n    \n\ndef cross_join(df):\n    df = df.drop(['datetime', 'play_id'], axis=1).merge(df.drop(['datetime', 'game_key', 'play_id'], axis=1), how='inner', on=['game_play', 'step'])\n    df['contact_id'] = df['game_play'] + '_' + df['step'].astype(str) + '_' + df['nfl_player_id_x'].astype(str) + '_' + df['nfl_player_id_y'].astype(str)\n    df = df[(df['nfl_player_id_x'] < df['nfl_player_id_y'])]\n    return df\n\ndef single_preprocess_add_suffix(df):\n    df = df.drop(['datetime'], axis=1)\n    df = pd.concat([df[['game_key', 'play_id', 'step', 'game_play']], df.drop(['game_key', 'play_id', 'game_play', 'step'], axis=1).add_suffix('_x')], axis=1)\n    return df\n\n\ndef create_conact_id_g(df):\n    df['contact_id'] = df['game_play'] + '_' + df['step'].astype(str) + '_' + df['nfl_player_id_x'].astype(str) + '_G'\n    return df\n\ndef create_features(df):\n    df['same_team'] = df['team_x'] == df['team_y']\n    return df\n\n\ndef calc_rolling_features(df, is_pair=False):\n    \n    if is_pair:\n        df = df.sort_values(['game_play', 'nfl_player_id_x', 'nfl_player_id_y', 'step']).reset_index(drop=True)\n        for shift in range(-10, 11):\n            df[f'player_distance_shift{shift}'] = df.groupby(['game_play', 'nfl_player_id_x', 'nfl_player_id_y'])['player_distance'].shift(shift)\n        df['player_distance_rolling_center_5_mean'] = df.groupby(['game_play', 'nfl_player_id_x', 'nfl_player_id_y'])['player_distance'].transform(lambda d: d.rolling(5, center=True).mean())\n        df['player_distance_rolling_center_5_std'] = df.groupby(['game_play', 'nfl_player_id_x', 'nfl_player_id_y'])['player_distance'].transform(lambda d: d.rolling(5, center=True).std())\n        #df['player_distance_rolling_5_mean'] = df['player_distance', 'nfl_player_id_x', 'nfl_player_id_y'].rolling(5).mean()\n    else:\n        df = df.sort_values(['game_play', 'nfl_player_id_x', 'step']).reset_index(drop=True)\n        \n    df['speed_x_rolling_center_3_mean'] = df.groupby(['game_play', 'nfl_player_id_x'])['speed_x'].transform(lambda d: d.rolling(3, center=True).mean())\n    df['speed_x_rolling_5_mean'] = df.groupby(['game_play', 'nfl_player_id_x'])['speed_x'].transform(lambda d: d.rolling(5).mean())\n\n    if is_pair:\n        df['speed_y_rolling_center_3_mean'] = df.groupby(['game_play', 'nfl_player_id_y'])['speed_y'].transform(lambda d: d.rolling(3, center=True).mean())\n        df['speed_y_rolling_5_mean'] = df.groupby(['game_play', 'nfl_player_id_y'])['speed_y'].transform(lambda d: d.rolling(5).mean())\n\n\n    df['acceleration_x_rolling_5_mean'] = df.groupby(['game_play', 'nfl_player_id_x'])['acceleration_x'].transform(lambda d: d.rolling(5).mean())\n\n    if is_pair:\n        df['acceleration_y_rolling_5_mean'] = df.groupby(['game_play', 'nfl_player_id_y'])['acceleration_y'].transform(lambda d: d.rolling(5).mean())\n\n    df['sa_x_rolling_5_mean'] = df.groupby(['game_play', 'nfl_player_id_x'])['sa_x'].transform(lambda d: d.rolling(5).mean())\n\n    if is_pair:\n        df['sa_y_rolling_5_mean'] = df.groupby(['game_play', 'nfl_player_id_y'])['sa_y'].transform(lambda d: d.rolling(5).mean())\n\n    return df\n\ndef cos_sim(v1, v2):\n    return np.dot(v1, v2) / (np.linalg.norm(v1) * np.linalg.norm(v2))\n\n\ndef calc_direction_features(df):\n\n    directions = []\n    orientations = []\n\n    span = 600_000\n    count = 0\n    while True:\n        _df = df.iloc[span*count:span*(count+1), :]\n\n        direction = _df.parallel_apply(lambda x : cos_sim([x['x_direction_x'], x['y_direction_x']], [x['x_direction_y'], x['y_direction_y']]), axis=1).to_numpy()\n        orientation = _df.parallel_apply(lambda x : cos_sim([x['x_orientation_x'], x['y_orientation_x']], [x['x_orientation_y'], x['y_orientation_y']]), axis=1).to_numpy()\n\n        directions.append(direction)\n        orientations.append(orientation)\n\n        count += 1\n\n        if span * count > len(df):\n            break\n\n    df['direction_cos_sim'] = np.hstack(directions)\n    df['orientation_cos_sim'] = np.hstack(orientations)\n\n    return df\n\ndef calc_player_distance(df):\n\tdf['player_distance'] = np.sqrt((df['x_position_x'] - df['x_position_y'])**2 + (df['y_position_x'] - df['y_position_y'])**2)\n\treturn df\n\ndef calc_frame(df):\n    df['frame'] = (df['step']/10*59.94+5*59.94).astype('int')+1\n    return df\n\n\ndef post_process(df):\n    df['play_id'] = df['game_play'].apply(lambda x : x.split('_')[1]).astype(int)\n    df = df.reset_index(drop=True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.923589Z","iopub.execute_input":"2023-02-16T15:22:06.923990Z","iopub.status.idle":"2023-02-16T15:22:06.963367Z","shell.execute_reply.started":"2023-02-16T15:22:06.923954Z","shell.execute_reply":"2023-02-16T15:22:06.962288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def facade_pair(_df, helmets):\n    #print(func)\n    df = calc_centroid(_df)\n    df = calc_direction(df)\n    df = concat_history(df)\n    df = calc_bbox_size(df, helmets)\n    df = cross_join(df)\n    df = create_features(df)\n    \n    df = calc_direction_features(df)\n    df = calc_player_distance(df)\n    df = calc_rolling_features(df, is_pair=True)\n    df = calc_frame(df)\n    df = post_process(df)\n\n    return df\n\ndef facade_g(_df, helmets):\n    #print(func)\n    df = calc_centroid(_df)\n    df = calc_direction(df)\n    df = concat_history(df)\n    df = calc_bbox_size(df, helmets)\n    #df = cross_join(df)\n    df = single_preprocess_add_suffix(df)\n    #df = create_features(df)\n    \n    #df = calc_direction_features(df)\n    #df = calc_player_distance(df)\n    df = create_conact_id_g(df)\n    df = calc_rolling_features(df, is_pair=False)\n    df = calc_frame(df)\n    df = post_process(df)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.964899Z","iopub.execute_input":"2023-02-16T15:22:06.965262Z","iopub.status.idle":"2023-02-16T15:22:06.973098Z","shell.execute_reply.started":"2023-02-16T15:22:06.965226Z","shell.execute_reply":"2023-02-16T15:22:06.971396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_features = ['y_position_shift_-1_y', 'x_position_shift_-1_x', 'x_position_team_centroid_x', 'x_position_centroid_x', 'player_max_acceleration_y', 'player_max_sa_y', 'y_position_y', 'y_position_shift_1_y', 'y_position_x', 'y_position_centroid_y', 'y_position_team_std_x', 'y_position_shift_1_x', 'y_position_std_x']\ndrop_features2 = ['orientation_shift_1_x', 'direction_shift_-1_y', 'orientation_y', 'direction_shift_1_x', 'orientation_shift_1_y', 'orientation_x']\ndrop_features3 = ['player_distance_team_centroid_y', 'x_direction_x', 'x_position_x', 'x_position_shift_1_x', 'y_orientation_x', 'y_position_team_centroid_y']\ndrop_features4 = ['y_direction_x', 'bbox_size_endzone_y', 'x_orientation_x', 'bbox_size_sideline_y', 'x_direction_y', 'direction_shift_1_y', 'player_max_speed_y', 'y_direction_y']\ndrop_features5 = ['direction_cos_sim', 'acceleration_shift_1_y', 'direction_y', 'direction_x', 'y_position_team_std_y']\ndrop_features6 = ['x_position_std_x', 'orientation_shift_-1_x', 'mean_bbox_size_sideline_y', 'orientation_shift_-1_y', 'orientation_cos_sim', 'acceleration_y', 'y_position_std_y', 'direction_shift_-1_x', 'sa_y', 'y_orientation_y', 'y_position_shift_-1_x']\n\ndrop_features = drop_features + drop_features2 + drop_features3 + drop_features4 + drop_features5 + drop_features6\n\n#if not is_kaggle:\n#    train_df = train_df.drop(drop_features, axis=1)\n#    \n#test_df = test_df.drop(drop_features, axis=1)\n\ndrop_features","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.975345Z","iopub.execute_input":"2023-02-16T15:22:06.976128Z","iopub.status.idle":"2023-02-16T15:22:06.988464Z","shell.execute_reply.started":"2023-02-16T15:22:06.976091Z","shell.execute_reply":"2023-02-16T15:22:06.987351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif not is_kaggle:\n    print('train')\n    train_pair_df = facade_pair(train_df, train_helmets)\n    train_pair_df['G_flag'] = False\n    train_g_df = facade_g(train_df, train_helmets)\n    train_g_df['G_flag'] = True\n\nprint('test')\ntest_pair_df = facade_pair(test_df, test_helmets)\ntest_pair_df['G_flag'] = False\ndrop_pair = set(test_pair_df.columns) & set(drop_features)\ntest_pair_df.drop(drop_pair, axis=1, inplace=True)\n\n\ntest_g_df = facade_g(test_df, test_helmets)\ntest_g_df['G_flag'] = True\ndrop_g = set(test_g_df.columns) & set(drop_features)\ntest_g_df.drop(drop_g, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:06.990250Z","iopub.execute_input":"2023-02-16T15:22:06.990701Z","iopub.status.idle":"2023-02-16T15:22:15.508424Z","shell.execute_reply.started":"2023-02-16T15:22:06.990664Z","shell.execute_reply":"2023-02-16T15:22:15.507214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    train_df = pd.concat([train_pair_df.reset_index(drop=True), train_g_df.reset_index(drop=True)], axis=0).reset_index(drop=True)\n    del train_pair_df, train_g_df\n\ntest_df  = pd.concat([test_pair_df.reset_index(drop=True), test_g_df.reset_index(drop=True)], axis=0).reset_index(drop=True)\ndel test_pair_df, test_g_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:15.510394Z","iopub.execute_input":"2023-02-16T15:22:15.511046Z","iopub.status.idle":"2023-02-16T15:22:15.624674Z","shell.execute_reply.started":"2023-02-16T15:22:15.510999Z","shell.execute_reply":"2023-02-16T15:22:15.623647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dict = {}","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:15.626293Z","iopub.execute_input":"2023-02-16T15:22:15.626684Z","iopub.status.idle":"2023-02-16T15:22:15.631969Z","shell.execute_reply.started":"2023-02-16T15:22:15.626647Z","shell.execute_reply":"2023-02-16T15:22:15.630514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n\n    data_count = 8000\n    train_labels_df['frame'] = (train_labels_df['step']/10*59.94+5*59.94).astype('int')+1\n    for idx, row in tqdm(train_labels_df[['game_play', 'step', 'frame']].drop_duplicates().head(data_count).iterrows(), total=data_count):\n        for v in ['Endzone', 'Sideline']:\n            file_name = f'{row[\"game_play\"]}_{v}.mp4_{row[\"frame\"]:04d}.jpg'\n\n            img = cv2.imread(f'../work/frames/{file_name}', 0)\n\n            if img is None:\n                continue\n\n            data_dict[file_name] = img\n        \nelse:\n    data_count = 2200\n    test_labels_df['frame'] = (test_labels_df['step']/10*59.94+5*59.94).astype('int')+1\n    for idx, row in tqdm(test_labels_df[['game_play', 'step', 'frame']].drop_duplicates().head(data_count).iterrows(), total=data_count):\n        for v in ['Endzone', 'Sideline']:\n            file_name = f'{row[\"game_play\"]}_{v}.mp4_{row[\"frame\"]:04d}.jpg'\n\n            img = cv2.imread(f'../work/frames/{file_name}', 0)\n\n            if img is None:\n                continue\n\n            data_dict[file_name] = img\n            \nprint(len(data_dict))","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:15.633911Z","iopub.execute_input":"2023-02-16T15:22:15.634342Z","iopub.status.idle":"2023-02-16T15:22:17.795560Z","shell.execute_reply.started":"2023-02-16T15:22:15.634308Z","shell.execute_reply":"2023-02-16T15:22:17.794459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    target = 'contact'\n    del_columns = ['contact', 'game_play', 'contact_id', 'team_x', 'position_x',\n                   'team_y', 'position_y', 'game_key', 'index_x', 'index_y', 'play_id', 'frame', 'nfl_player_id_x', 'nfl_player_id_y']\n\n    features = list(set(train_df.columns) - set(del_columns))\n    features.sort()\n    print(features)\n    print(len(features))\n\n    to_pickle(f'../output/{HOST}_{NB}_features.pkl', features)\n    \nelse:\n    features = unpickle(f'/kaggle/input/nfl-models/{HOST}_{NB}_features.pkl')\n    \nfeatures","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:17.797453Z","iopub.execute_input":"2023-02-16T15:22:17.798760Z","iopub.status.idle":"2023-02-16T15:22:17.813992Z","shell.execute_reply.started":"2023-02-16T15:22:17.798703Z","shell.execute_reply":"2023-02-16T15:22:17.812826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 件数絞る","metadata":{}},{"cell_type":"code","source":"\nif not is_kaggle:\n    \n    # 後処理用特徴量\n    pp_feature_train = train_df[['contact_id', 'speed_x', 'speed_y', 'player_distance', 'G_flag', 'step', 'same_team']].copy()\n    \n    train_df = train_df[train_df['contact_id'].isin(train_labels_df['contact_id'])]\n\n    train_hight_distance_df = train_df[train_df['player_distance']  > 2.5]\n    train_df = train_df[~(train_df['player_distance']  > 2.5)].reset_index(drop=True)\n    train_labels_df = train_labels_df[train_labels_df['contact_id'].isin(train_df['contact_id'])].reset_index(drop=True)\n\n    train_df = train_df.reset_index(drop=True)\n    train_labels_df = train_labels_df.reset_index(drop=True)\n    \n    # 不要特徴量削除\n    train_df = train_df.drop(['team_x', 'position_x', 'team_y', 'position_y'], axis=1)\n    \n    # ラベルデータに存在するもののみ抽出\n    _label = pd.read_csv(DATA_DIR / Path('train_labels.csv'))\n    train_df = train_df.merge(_label[['contact_id', 'contact']], how='left', on='contact_id')\n    train_hight_distance_df = train_hight_distance_df.merge(_label[['contact_id', 'contact']], how='left', on='contact_id')\n    \n\n\n# 後処理用特徴量\npp_feature_test = test_df[['contact_id', 'speed_x', 'speed_y', 'player_distance', 'G_flag', 'step', 'same_team']].copy()\n\ntest_df = test_df[test_df['contact_id'].isin(test_labels_df['contact_id'])]\n\ntest_hight_distance_df = test_df[test_df['player_distance']  > 2.5]\ntest_df = test_df[~(test_df['player_distance']  > 2.5)].reset_index(drop=True)\ntest_labels_df = test_labels_df[test_labels_df['contact_id'].isin(test_df['contact_id'])].reset_index(drop=True)\n\ntest_df = test_df.reset_index(drop=True)\ntest_labels_df = test_labels_df.reset_index(drop=True)\n\n# 不要特徴量削除\ntest_df = test_df.drop(['team_x', 'position_x', 'team_y', 'position_y'], axis=1)\n\n# 提出対象のみ抽出\n_label = pd.read_csv(DATA_DIR / Path('sample_submission.csv'))\ntest_df = test_df.merge(_label[['contact_id', 'contact']], how='left', on='contact_id')\ntest_hight_distance_df = test_hight_distance_df.merge(_label[['contact_id', 'contact']], how='left', on='contact_id')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:17.815706Z","iopub.execute_input":"2023-02-16T15:22:17.816198Z","iopub.status.idle":"2023-02-16T15:22:18.096371Z","shell.execute_reply.started":"2023-02-16T15:22:17.816154Z","shell.execute_reply":"2023-02-16T15:22:18.095349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del _label\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.099083Z","iopub.execute_input":"2023-02-16T15:22:18.099462Z","iopub.status.idle":"2023-02-16T15:22:18.303857Z","shell.execute_reply.started":"2023-02-16T15:22:18.099425Z","shell.execute_reply":"2023-02-16T15:22:18.302615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### YOLO","metadata":{}},{"cell_type":"code","source":"## ファイルパス\n\nif not is_kaggle:\n    yolo_model_dir = \"/home/yuki/ml/kaggle/nfl-player-contact-detection/git/yolov5\"\n    yolo_model_path = '/home/yuki/ml/kaggle/nfl-player-contact-detection/git/yolov5/' + Config.YOLO_SIZE\n    file_dir = '/home/yuki/ml/kaggle/nfl-player-contact-detection/work/frames/'\nelse:\n    yolo_model_dir = '/kaggle/input/yoloall'\n    yolo_model_path = '/kaggle/input/yoloall/' + Config.YOLO_SIZE\n    file_dir = '/kaggle/work/frames/'","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.305627Z","iopub.execute_input":"2023-02-16T15:22:18.305926Z","iopub.status.idle":"2023-02-16T15:22:18.312255Z","shell.execute_reply.started":"2023-02-16T15:22:18.305899Z","shell.execute_reply":"2023-02-16T15:22:18.311253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 存在しないファイルを見つける\nendzone_not_exist_files = []\n\nif not is_kaggle:\n    for f in tqdm((file_dir + train_df['game_play'] + '_Endzone.mp4_' + train_df['frame'].apply(lambda x : f'{x:04}') + '.jpg').unique()):\n        if not os.path.exists(f):        \n            endzone_not_exist_files.append(f)\n\nfor f in tqdm((file_dir + test_df['game_play'] + '_Endzone.mp4_' + test_df['frame'].apply(lambda x : f'{x:04}') + '.jpg').unique()):\n    if not os.path.exists(f):        \n        endzone_not_exist_files.append(f)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.313673Z","iopub.execute_input":"2023-02-16T15:22:18.314693Z","iopub.status.idle":"2023-02-16T15:22:18.338000Z","shell.execute_reply.started":"2023-02-16T15:22:18.314654Z","shell.execute_reply":"2023-02-16T15:22:18.336944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    endzone_train_helemts = train_helmets[train_helmets['view'] == 'Endzone'][['game_play', 'frame']]\n    sideline_train_helemts = train_helmets[train_helmets['view'] == 'Sideline'][['game_play', 'frame']]\n    game_play_max_frame = endzone_train_helemts.merge(sideline_train_helemts, how='inner', on=['game_play', 'frame']).groupby('game_play')['frame'].max()\nelse:\n    endzone_test_helemts = test_helmets[test_helmets['view'] == 'Endzone'][['game_play', 'frame']]\n    sideline_test_helemts = test_helmets[test_helmets['view'] == 'Sideline'][['game_play', 'frame']]\n    game_play_max_frame = endzone_test_helemts.merge(sideline_test_helemts, how='inner', on=['game_play', 'frame']).groupby('game_play')['frame'].max()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.339607Z","iopub.execute_input":"2023-02-16T15:22:18.339948Z","iopub.status.idle":"2023-02-16T15:22:18.411959Z","shell.execute_reply.started":"2023-02-16T15:22:18.339914Z","shell.execute_reply":"2023-02-16T15:22:18.410967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#直近のframe調べる\n\nendzone_file_df = pd.DataFrame(endzone_not_exist_files, columns=['full_path']).drop_duplicates().reset_index(drop=True)\nendzone_file_df['file_name'] = endzone_file_df['full_path'].apply(lambda x : x.split('/')[-1])\nendzone_file_df['game_play'] = endzone_file_df['file_name'].apply(lambda x : x[:12])\nendzone_file_df['frame'] = endzone_file_df['file_name'].apply(lambda x : x[:29][25:]).astype(int)\n\nnew_frame = []\n\nfor idx, row in tqdm(endzone_file_df.iterrows()):\n    file_frame = row['frame']\n    \n    max_frame = game_play_max_frame.loc[row['game_play']]\n    new_frame.append(max_frame)\n            \nendzone_file_df['new_frame'] = new_frame\nendzone_file_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.413602Z","iopub.execute_input":"2023-02-16T15:22:18.413982Z","iopub.status.idle":"2023-02-16T15:22:18.440088Z","shell.execute_reply.started":"2023-02-16T15:22:18.413942Z","shell.execute_reply":"2023-02-16T15:22:18.438923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n(endzone_file_df['frame'] - endzone_file_df['new_frame']).hist(bins=25)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.441469Z","iopub.execute_input":"2023-02-16T15:22:18.441919Z","iopub.status.idle":"2023-02-16T15:22:18.712018Z","shell.execute_reply.started":"2023-02-16T15:22:18.441884Z","shell.execute_reply":"2023-02-16T15:22:18.711058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    for idx, row in endzone_file_df.iterrows():\n        train_df.loc[(train_df['game_play'] == row['game_play']) & (train_df['frame'] == row['frame']), 'frame'] = row['new_frame']\n        #train_helmets.loc[(train_helmets['game_play'] == row['game_play']) & (train_helmets['frame'] == row['frame']), 'frame'] = row['new_frame']\n        \n    \nfor idx, row in endzone_file_df.iterrows():\n    test_df.loc[(test_df['game_play'] == row['game_play']) & (test_df['frame'] == row['frame']), 'frame'] = row['new_frame']\n    #test_helmets.loc[(test_helmets['game_play'] == row['game_play']) & (test_helmets['frame'] == row['frame']), 'frame'] = row['new_frame']","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.713377Z","iopub.execute_input":"2023-02-16T15:22:18.713980Z","iopub.status.idle":"2023-02-16T15:22:18.722928Z","shell.execute_reply.started":"2023-02-16T15:22:18.713941Z","shell.execute_reply":"2023-02-16T15:22:18.721905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#YOLOには関係ない。ヘルメット検出あるかどうか。\ndef exit_helmet(df, helmets, view):\n    _helmets = helmets[helmets['view'] == view]\n    train_x_exist = df.merge(_helmets[['game_play', 'nfl_player_id', 'frame']].drop_duplicates(), how='inner', left_on=['game_play', 'nfl_player_id_x', 'frame'], right_on=['game_play', 'nfl_player_id', 'frame'])[['contact_id']]\n    train_y_exist = df.merge(_helmets[['game_play', 'nfl_player_id', 'frame']].drop_duplicates(), how='inner', left_on=['game_play', 'nfl_player_id_y', 'frame'], right_on=['game_play', 'nfl_player_id', 'frame'])[['contact_id']]\n    df[f'exist_x_helmet_{view}'] = df['contact_id'].isin(train_x_exist['contact_id'])\n    df[f'exist_y_helmet_{view}'] = df['contact_id'].isin(train_y_exist['contact_id'])\n    df[f'exist_both_helmet_{view}'] = df[f'exist_x_helmet_{view}'] * df[f'exist_y_helmet_{view}']\n    return df\n\nif not is_kaggle:\n    train_df = exit_helmet(train_df, train_helmets, 'Endzone')\n    train_df = exit_helmet(train_df, train_helmets, 'Sideline')\n\ntest_df = exit_helmet(test_df, test_helmets, 'Endzone')\ntest_df = exit_helmet(test_df, test_helmets, 'Sideline')","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.724267Z","iopub.execute_input":"2023-02-16T15:22:18.725989Z","iopub.status.idle":"2023-02-16T15:22:18.823231Z","shell.execute_reply.started":"2023-02-16T15:22:18.725875Z","shell.execute_reply":"2023-02-16T15:22:18.822252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# YOLOに通すファイルパス一覧取得\n\nexist_files = []\n\nif not is_kaggle:\n    for f in tqdm((file_dir + train_df['game_play'] + '_Endzone.mp4_' + train_df['frame'].apply(lambda x : f'{x:04}') + '.jpg').unique()):\n        if os.path.exists(f):        \n            exist_files.append(f)\n    print(len(exist_files))\n        \n    for f in tqdm((file_dir + train_df['game_play'] + '_Sideline.mp4_' + train_df['frame'].apply(lambda x : f'{x:04}') + '.jpg').unique()):\n        if os.path.exists(f):        \n            exist_files.append(f)        \n    print(len(exist_files))\n\nfor f in tqdm((file_dir + test_df['game_play'] + '_Endzone.mp4_' + test_df['frame'].apply(lambda x : f'{x:04}') + '.jpg').unique()):\n    if os.path.exists(f):        \n        exist_files.append(f)\nfor f in tqdm((file_dir + test_df['game_play'] + '_Sideline.mp4_' + test_df['frame'].apply(lambda x : f'{x:04}') + '.jpg').unique()):\n    if os.path.exists(f):        \n        exist_files.append(f)\n\n#results = model(['/home/yuki/ml/kaggle/nfl-player-contact-detection/work/frames/' + f])","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.833470Z","iopub.execute_input":"2023-02-16T15:22:18.833778Z","iopub.status.idle":"2023-02-16T15:22:18.871098Z","shell.execute_reply.started":"2023-02-16T15:22:18.833750Z","shell.execute_reply":"2023-02-16T15:22:18.870146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yolo_bbox_dict = {}\n\nmodel = torch.hub.load(yolo_model_dir, 'custom', path=yolo_model_path, source='local', device=0)\n\nn = 10\nfor i in tqdm(range(0, len(exist_files), n)):\n    result = model(exist_files[i: i+n])\n    \n    xyxy = result.pandas().xyxy\n    \n    for j, f in enumerate(exist_files[i: i+n]):\n        _df = xyxy[j]\n        _df = _df[(_df['class'] == 0) & (_df['confidence'] > 0.2)]\n\n        \"\"\"_df['left'] = (_df['xmin'] - (_df['xmax'] / 2)).astype(int)\n        _df['width'] = _df['xmax'].astype(int)\n        _df['top'] = (_df['ymin'] - (_df['ymax'] / 2)).astype(int)\n        _df['height'] = _df['ymax'].astype(int)\"\"\"\n        \n        _df['left'] = _df['xmin'].astype(int)\n        _df['width'] = (_df['xmax'] -_df['xmin']).astype(int)\n        _df['top'] = _df['ymin'].astype(int)\n        _df['height'] = (_df['ymax'] -_df['ymin']).astype(int)\n        \n        _df = _df.drop(['xmin', 'xmax', 'ymin', 'ymax', 'name', 'class', 'confidence'], axis=1)\n        \n        yolo_bbox_dict[f.split('/')[-1]] = _df\n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:18.872309Z","iopub.execute_input":"2023-02-16T15:22:18.872649Z","iopub.status.idle":"2023-02-16T15:22:45.973774Z","shell.execute_reply.started":"2023-02-16T15:22:18.872614Z","shell.execute_reply":"2023-02-16T15:22:45.972624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    to_pickle(f'../output/{HOST}_{NB}_yolo_bbox_dict_yolov5x6.pkl', yolo_bbox_dict)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:45.975502Z","iopub.execute_input":"2023-02-16T15:22:45.975917Z","iopub.status.idle":"2023-02-16T15:22:45.983234Z","shell.execute_reply.started":"2023-02-16T15:22:45.975879Z","shell.execute_reply":"2023-02-16T15:22:45.982209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del model, result, xyxy, _df, endzone_file_df\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:45.984657Z","iopub.execute_input":"2023-02-16T15:22:45.985343Z","iopub.status.idle":"2023-02-16T15:22:46.166622Z","shell.execute_reply.started":"2023-02-16T15:22:45.985298Z","shell.execute_reply":"2023-02-16T15:22:46.165583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 標準化","metadata":{}},{"cell_type":"code","source":"if not is_kaggle:\n    ss = StandardScaler()\nelse:\n    ss = unpickle(f'/kaggle/input/nfl-models/{HOST}_{NB}_ss.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.168329Z","iopub.execute_input":"2023-02-16T15:22:46.168715Z","iopub.status.idle":"2023-02-16T15:22:46.180990Z","shell.execute_reply.started":"2023-02-16T15:22:46.168679Z","shell.execute_reply":"2023-02-16T15:22:46.179883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nif not is_kaggle:\n    train_df = pd.concat([train_df.drop(features, axis=1), pd.DataFrame(ss.fit_transform(train_df[features]), columns=features)], axis=1)\n    train_df[(train_df['contact_id'].isin(train_labels_df['contact_id']))]\n    train_df = train_df.merge(train_labels_df[['contact_id']], how='inner', on='contact_id')\n\n    train_labels_df = train_labels_df.sort_values('contact_id').reset_index(drop=True)\n    train_df = train_df.sort_values('contact_id').reset_index(drop=True)\n    train_df = train_df.fillna(-1)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.182416Z","iopub.execute_input":"2023-02-16T15:22:46.182802Z","iopub.status.idle":"2023-02-16T15:22:46.191611Z","shell.execute_reply.started":"2023-02-16T15:22:46.182768Z","shell.execute_reply":"2023-02-16T15:22:46.188516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsample_submission_df['nfl_player_id_2'] = sample_submission_df['contact_id'].apply(lambda x : x.split('_')[-1])\ntest_df = test_df.merge(sample_submission_df, how='inner', on='contact_id')\n\nsample_submission_df = sample_submission_df.sort_values('contact_id').reset_index(drop=True)\ntest_df = test_df.sort_values('contact_id').reset_index(drop=True)\n\ntest_df = pd.concat([test_df.drop(features, axis=1), pd.DataFrame(ss.transform(test_df[features]), columns=features)], axis=1)\ntest_df = test_df.fillna(-1)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.193163Z","iopub.execute_input":"2023-02-16T15:22:46.193523Z","iopub.status.idle":"2023-02-16T15:22:46.386442Z","shell.execute_reply.started":"2023-02-16T15:22:46.193488Z","shell.execute_reply":"2023-02-16T15:22:46.385414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    to_pickle(f'../output/{HOST}_{NB}_ss.pkl', ss)\ndel ss","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.388640Z","iopub.execute_input":"2023-02-16T15:22:46.389331Z","iopub.status.idle":"2023-02-16T15:22:46.394444Z","shell.execute_reply.started":"2023-02-16T15:22:46.389290Z","shell.execute_reply":"2023-02-16T15:22:46.393414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"train_df = train_df.sort_values(['game_play', 'nfl_player_id_x', 'nfl_player_id_y', 'step'])\ntrain_df['contact_shift1'] = train_df.groupby(['game_play', 'nfl_player_id_x', 'nfl_player_id_y'])['contact'].shift(1).fillna(method='ffill').fillna(method='bfill')\ntrain_df['contact_shift-1'] = train_df.groupby(['game_play', 'nfl_player_id_x', 'nfl_player_id_y'])['contact'].shift(-1).fillna(method='ffill').fillna(method='bfill')\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.395971Z","iopub.execute_input":"2023-02-16T15:22:46.396343Z","iopub.status.idle":"2023-02-16T15:22:46.406161Z","shell.execute_reply.started":"2023-02-16T15:22:46.396308Z","shell.execute_reply":"2023-02-16T15:22:46.405076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video2helmets = {}\nvideo2frames = {}","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.407678Z","iopub.execute_input":"2023-02-16T15:22:46.408016Z","iopub.status.idle":"2023-02-16T15:22:46.414908Z","shell.execute_reply.started":"2023-02-16T15:22:46.407983Z","shell.execute_reply":"2023-02-16T15:22:46.413893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nif not is_kaggle:\n    train_helmets_new = train_helmets[['video', 'nfl_player_id', 'frame', 'left', 'width', 'top', 'height']].copy().set_index('video')\n    for video in tqdm(train_helmets.video.unique()):\n        video2helmets[video] = train_helmets_new.loc[video].reset_index(drop=True)\n\n    del train_helmets_new\n    gc.collect()\n\n    for game_play in tqdm(train_video_metadata.game_play.unique()):\n        for view in ['Endzone', 'Sideline']:\n            video = game_play + f'_{view}.mp4'\n            video2frames[video] = max(list(map(lambda x:int(x.split('_')[-1].split('.')[0]), glob.glob(f'../work/frames/{video}*'))))\n\nelse:\n    test_helmets_new = test_helmets[['video', 'nfl_player_id', 'frame', 'left', 'width', 'top', 'height']].copy().set_index('video')\n    for video in tqdm(test_helmets.video.unique()):\n        video2helmets[video] = test_helmets_new.loc[video].reset_index(drop=True)\n\n    del test_helmets_new\n    gc.collect()\n\n    for game_play in tqdm(test_video_metadata.game_play.unique()):\n        for view in ['Endzone', 'Sideline']:\n            video = game_play + f'_{view}.mp4'\n            video2frames[video] = max(list(map(lambda x:int(x.split('_')[-1].split('.')[0]), glob.glob(f'../work/frames/{video}*'))))","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.416397Z","iopub.execute_input":"2023-02-16T15:22:46.416755Z","iopub.status.idle":"2023-02-16T15:22:46.631186Z","shell.execute_reply.started":"2023-02-16T15:22:46.416719Z","shell.execute_reply":"2023-02-16T15:22:46.630228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.632912Z","iopub.execute_input":"2023-02-16T15:22:46.633636Z","iopub.status.idle":"2023-02-16T15:22:46.779787Z","shell.execute_reply.started":"2023-02-16T15:22:46.633599Z","shell.execute_reply":"2023-02-16T15:22:46.778617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_img_single(frame, video, players, window=24, frame_interval=1, crop_size=128, agg_frame=True):\n    \n    crop_size_half = int(crop_size / 2)\n    \n    imgs = []\n\n    tmp = video2helmets[video]\n    tmp = tmp[(tmp['frame'].between(frame-window, frame+window)) & (tmp.nfl_player_id.isin(players))]\n    #tmp = tmp[tmp.nfl_player_id.isin(players)]#.sort_values(['nfl_player_id', 'frame'])\n    #tmp_player = tmp.interpolate(limit_direction='both').copy()\n    #tmp_frames = tmp.frame.values\n    \n    zero_img = np.zeros((720, 1280), dtype=np.float32)\n        \n    for idx, player_gp_df in tmp.groupby('nfl_player_id'):\n        \n        if len(player_gp_df) <= window * 2:\n            player_gp_df = player_gp_df.reset_index()\n            helmet_bbox = pd.DataFrame([i for i in range(frame - window, frame+window + 1)], columns=['frame']).merge(player_gp_df, how='left', on='frame').interpolate(limit_direction='both').copy().iloc[window]\n        else:\n            helmet_bbox = player_gp_df.iloc[window]\n        \n        #helmet_bbox = player_gp_df.interpolate(limit_direction='both')[['left','width','top','height']].iloc[window]\n        zero_img = cv2.rectangle(zero_img, (int(helmet_bbox['left']), int(helmet_bbox['top'])), (int(helmet_bbox['left'] + helmet_bbox['width']), int(helmet_bbox['top'] + helmet_bbox['height'])), (255, 255, 255))\n        \n    \n    \n    if agg_frame:\n        tmp = tmp.groupby('frame')[['left','width','top','height']].mean()\n    \n    if len(tmp) <= window * 2:\n        tmp = tmp.reset_index()\n        bboxes = pd.DataFrame([i for i in range(frame - window, frame+window + 1)], columns=['frame']).merge(tmp, how='left', on='frame').interpolate(limit_direction='both').copy().to_numpy()[: ,1:][::frame_interval]\n    else:\n        bboxes = tmp.to_numpy()[::frame_interval]\n        \n        \n    if bboxes.sum() > 0:\n        flag = 1\n    else:\n        flag = 0\n\n    img_new = np.zeros((crop_size, crop_size), dtype=np.float32)\n    img_zero_new = np.zeros((crop_size, crop_size), dtype=np.float32)\n    \n    \n    if flag == 1 and frame <= video2frames[video]:\n\n        if f'{video}_{frame:04d}.jpg' in  data_dict:\n            img = data_dict[f'{video}_{frame:04d}.jpg']\n        else:\n            img = cv2.imread(f'../work/frames/{video}_{frame:04d}.jpg', 0)\n        \n        #print(len(bboxes))\n        x, w, y, h = bboxes[window]\n\n        img = img[int(y+h/2)-crop_size_half:int(y+h/2)+crop_size_half,int(x+w/2)-crop_size_half:int(x+w/2)+crop_size_half].copy()\n        zero_img = zero_img[int(y+h/2)-crop_size_half:int(y+h/2)+crop_size_half,int(x+w/2)-crop_size_half:int(x+w/2)+crop_size_half].copy()\n        \n        \n\n        img_new[:img.shape[0], :img.shape[1]] = img\n        img_zero_new[:zero_img.shape[0], :zero_img.shape[1]] = zero_img\n\n    imgs.append(img_new)\n    imgs.append(img_zero_new)\n\n    return np.array(imgs)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.782010Z","iopub.execute_input":"2023-02-16T15:22:46.782303Z","iopub.status.idle":"2023-02-16T15:22:46.800478Z","shell.execute_reply.started":"2023-02-16T15:22:46.782277Z","shell.execute_reply":"2023-02-16T15:22:46.799248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    fig, axes = plt.subplots(4,6,figsize=(30, 25))\n    axes = axes.reshape(-1)\n    _train_df = train_df[train_df['G_flag'] < 0]\n    for i, j in enumerate(range(0, 24, 2)):\n        row = _train_df.iloc[j * 300]\n        #print(row['frame'], f'{row[\"game_play\"]}_Endzone.mp4', [row['nfl_player_id_x'], row['nfl_player_id_y']])\n        img=create_img_single(row['frame'], f'{row[\"game_play\"]}_Sideline.mp4', [row['nfl_player_id_x'], row['nfl_player_id_y']])\n        axes[j].set_title(f'{row[\"game_play\"]}_Endzone.mp4')\n        axes[j].imshow(img[0,:,:])\n        axes[j+1].imshow(img[1,:,:])","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.802303Z","iopub.execute_input":"2023-02-16T15:22:46.803055Z","iopub.status.idle":"2023-02-16T15:22:46.812147Z","shell.execute_reply.started":"2023-02-16T15:22:46.803017Z","shell.execute_reply":"2023-02-16T15:22:46.811108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_img_single_yolo(frame, video, players, window=24, agg_frame=True, is_debug=False):\n    \n    \n    imgs = [np.zeros((100, 100), dtype=np.float32), np.zeros((100, 100), dtype=np.float32)]\n    \n    \n    if frame > video2frames[video]:\n        #print(1)\n        return np.array(imgs)\n    \n    \n    bbox_filename = f'{video}_{frame:04}.jpg'\n    \n    #if not os.path.exists(f'/kaggle/temp/yolo/yolov5main/runs/detect/{video}/labels/{bbox_filename}'):\n    if not bbox_filename in yolo_bbox_dict:\n        #print(2)\n        return np.array(imgs)\n        \n    zero_img = np.zeros((720, 1280), dtype=np.float32)\n    yolo_bbox_df = yolo_bbox_dict[bbox_filename]\n    \n    #yolo_bbox_df = yolo_bbox_df[yolo_bbox_df[5] > 0.15].reset_index(drop=True)\n    \n    if len(players) <= 0:\n        raise Error\n\n    imgs = []\n\n    tmp = video2helmets[video]\n    \n    #display(tmp)\n    \n    tmp = tmp[(tmp['frame'].between(frame-window, frame+window))]\n    \n    \n    \n    \n    \"\"\"if f'{video}_{frame:04d}.jpg' in  data_dict:\n        img = data_dict[f'{video}_{frame:04d}.jpg']\n    else:\n        img = cv2.imread(f'../work/frames/{video}_{frame:04d}.jpg', 0)\"\"\"\n    \n    \n    # プレイヤーごとに、まるっと収まるbboxがぞんざいするか調べる\n    lefts = []\n    rights = []\n    tops = []\n    bottoms = []\n    \n    yolo_bboxes = []\n    for  player in players:\n        \n        player_tmp = tmp[tmp['nfl_player_id'] == player]\n        if len(player_tmp) <= 0:\n            continue\n                \n        #print('==')\n        #display(player_tmp)\n        player_tmp = player_tmp.merge(pd.DataFrame([ i for i in range(frame-window, frame+window + 1)], columns=['frame']), how='right')\n        \n        \n        player_tmp = player_tmp.interpolate(limit_direction='both').copy().fillna(-1).astype(int)\n        #display(len(player_tmp))\n        #display(player_tmp)        \n        \n        \n        \n        player_tmp = player_tmp[player_tmp['frame'] == frame]\n        \n        \n        \n        \n        if len(player_tmp) <= 0:\n            continue\n            \n        #print(player)\n                \n        player_tmp = player_tmp.iloc[0]\n        \n        if player_tmp.min() == -1:\n            continue\n            \n        \n\n        #img = cv2.rectangle(img, (player_tmp['left'], player_tmp['top']), (player_tmp['left'] + player_tmp['width'], player_tmp['top'] + player_tmp['height']), (255, 255, 255))\n        \n                \n        player_yolo_bbox = yolo_bbox_df[(yolo_bbox_df['left'] <= player_tmp['left']) & \n                                        ((yolo_bbox_df['left'] + yolo_bbox_df['width']) >= (player_tmp['left'] + player_tmp['width'])) & \n                                        (yolo_bbox_df['top'] <= player_tmp['top']) & \n                                        ((yolo_bbox_df['top'] + yolo_bbox_df['height']) >= (player_tmp['top'] + player_tmp['height']))]\n    \n        if len(player_yolo_bbox) <= 0:\n            \n            if 'Endzone' in video:\n            \n                #left\n                left = player_tmp['left'] + (player_tmp['width'] / 2) - (player_tmp['width'] * 1.9027794127672524)\n                right = player_tmp['left'] + (player_tmp['width'] / 2) + (player_tmp['width'] * 1.8881855717237415)\n                top = player_tmp['top'] + (player_tmp['height'] / 2) - (player_tmp['height'] * 1.2480045915236708)\n                bottom = player_tmp['top'] + (player_tmp['height'] / 2) + (player_tmp['height'] * 4.978755429664949)\n            else:\n                left = player_tmp['left'] + (player_tmp['width'] / 2) - (player_tmp['width'] * 1.7274801216452258)\n                right = player_tmp['left'] + (player_tmp['width'] / 2) + (player_tmp['width'] * 1.7640966198325152)\n                top = player_tmp['top'] + (player_tmp['height'] / 2) - (player_tmp['height'] * 1.1827818130570544)\n                bottom = player_tmp['top'] + (player_tmp['height'] / 2) + (player_tmp['height'] * 4.395735812212655)\n\n            lefts.append(int(left))\n            rights.append(int(right))\n            tops.append(int(top))\n            bottoms.append(int(bottom))\n\n        else:\n            lefts.append(player_yolo_bbox['left'].min())\n            rights.append((player_yolo_bbox['left'] + player_yolo_bbox['width']).max())\n            tops.append(player_yolo_bbox['top'].min())\n            bottoms.append((player_yolo_bbox['top'] + player_yolo_bbox['height']).max())\n            \n        zero_img = cv2.rectangle(zero_img, (player_tmp['left'], player_tmp['top']), (player_tmp['left'] + player_tmp['width'], player_tmp['top'] + player_tmp['height']), (255, 255, 255))\n\n    if len(lefts) > 0:\n        flag = 1\n    else:\n        flag = 0\n\n    if flag == 1 and frame <= video2frames[video]:\n        \n        if f'{video}_{frame:04d}.jpg' in  data_dict:\n            img = data_dict[f'{video}_{frame:04d}.jpg']\n        else:\n            img = cv2.imread(f'../work/frames/{video}_{frame:04d}.jpg', 0)\n\n        img = img[min(tops):max(bottoms), min(lefts):max(rights)]\n        zero_img = zero_img[min(tops):max(bottoms), min(lefts):max(rights)]\n    else:\n        #print(3)\n        return np.array([np.zeros((100, 100), dtype=np.float32), np.zeros((100, 100), dtype=np.float32)])\n    \n    if img.shape[0] == 0 or img.shape[1] == 0:\n        #return np.array(imgs)\n        #print(4)\n        return np.array([np.zeros((100, 100), dtype=np.float32), np.zeros((100, 100), dtype=np.float32)])\n\n    images = []\n    images.append(img)\n    images.append(zero_img)\n    \n    return np.array(images)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.813866Z","iopub.execute_input":"2023-02-16T15:22:46.814489Z","iopub.status.idle":"2023-02-16T15:22:46.840180Z","shell.execute_reply.started":"2023-02-16T15:22:46.814451Z","shell.execute_reply":"2023-02-16T15:22:46.838951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nif not is_kaggle:\n    fig, axes = plt.subplots(16,6,figsize=(30, 80))\n    axes = axes.reshape(-1)\n    _train_df = train_df[train_df['G_flag'] < 0]\n    \n    for i, j in enumerate(range(0, 96, 2)):\n        #print(j)\n        row = _train_df.iloc[j * 3500]\n        #print(row['frame'], f'{row[\"game_play\"]}_Endzone.mp4', [row['nfl_player_id_x'], row['nfl_player_id_y']])\n        img=create_img_single_yolo(row['frame'], f'{row[\"game_play\"]}_Sideline.mp4', [row['nfl_player_id_x'], row['nfl_player_id_y']], window=1000)\n        #print(row['frame'], f'{row[\"game_play\"]}_Sideline.mp4', [row['nfl_player_id_x'], row['nfl_player_id_y']])\n        axes[j].set_title(f'{row[\"game_play\"]}_Endzone.mp4_{row[\"frame\"]}')\n        axes[j].imshow(img[0,:,:])\n        axes[j+1].imshow(img[1,:,:])","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.842573Z","iopub.execute_input":"2023-02-16T15:22:46.843477Z","iopub.status.idle":"2023-02-16T15:22:46.856002Z","shell.execute_reply.started":"2023-02-16T15:22:46.843439Z","shell.execute_reply":"2023-02-16T15:22:46.854835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_augmentation():\n    train_transform = [\n        albu.HorizontalFlip(p=0.5),\n        #albu.ShiftScaleRotate(p=0.5),\n        albu.RandomBrightnessContrast(brightness_limit=(-0.3, 0.3), contrast_limit=(-0.3, 0.3), p=0.5),\n        #albu.VerticalFlip(p=0.5),\n        #albu.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.3, p=0.5),\n        #albu.LongestMaxSize(max_size=Config.IMG_SIZE, always_apply=False),\n        #albu.RandomResizedCrop(p=0.5, scale=[0.9, 1.0], height=Config.IMG_SIZE, width=Config.IMG_SIZE),\n        albu.ShiftScaleRotate(p=0.5, shift_limit=0.2, scale_limit=0.3, rotate_limit=4, border_mode=0, value=0, mask_value=0),\n        \n        #albu.Cutout(num_holes=3, max_h_size=30, max_w_size=30, fill_value=0, p=0.5),\n        \n        albu.Resize(height=Config.IMG_SIZE, width=Config.IMG_SIZE),\n        #albu.PadIfNeeded(always_apply=True, min_height=Config.IMG_SIZE, min_width=Config.IMG_SIZE, border_mode=2),\n        albu.Normalize(mean=(0), std=(1)),\n        #albu.ToSepia(p=0.3),\n        #albu.ToGray(p=1)\n    ]\n    return albu.Compose(train_transform)\n\n\ndef get_soft_augmentation():\n    train_transform = [\n        albu.HorizontalFlip(p=0.5),\n        #albu.ShiftScaleRotate(p=0.5),\n        albu.RandomBrightnessContrast(brightness_limit=(-0.1, 0.1), contrast_limit=(-0.1, 0.1), p=0.3),\n        #albu.VerticalFlip(p=0.5),\n        #albu.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.3, p=0.5),\n        #albu.LongestMaxSize(max_size=Config.IMG_SIZE, always_apply=False),\n        #albu.RandomResizedCrop(p=0.5, scale=[0.9, 1.0], height=Config.IMG_SIZE, width=Config.IMG_SIZE),\n        albu.ShiftScaleRotate(p=0.3, shift_limit=0.1, scale_limit=0.1, rotate_limit=1, border_mode=0, value=0, mask_value=0),\n        \n        #albu.Cutout(num_holes=3, max_h_size=30, max_w_size=30, fill_value=0, p=0.5),\n        \n        albu.Resize(height=Config.IMG_SIZE, width=Config.IMG_SIZE),\n        #albu.PadIfNeeded(always_apply=True, min_height=Config.IMG_SIZE, min_width=Config.IMG_SIZE, border_mode=2),\n        albu.Normalize(mean=(0), std=(1)),\n        #albu.ToSepia(p=0.3),\n        #albu.ToGray(p=1)\n    ]\n    return albu.Compose(train_transform)\n\n\ndef get_test_augmentation():\n    train_transform = [\n        albu.Resize(height=Config.IMG_SIZE, width=Config.IMG_SIZE),\n        albu.Normalize(mean=(0), std=(1)),\n    ]\n    return albu.Compose(train_transform)","metadata":{"papermill":{"duration":0.033222,"end_time":"2021-09-27T02:55:08.979344","exception":false,"start_time":"2021-09-27T02:55:08.946122","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:46.859420Z","iopub.execute_input":"2023-02-16T15:22:46.859788Z","iopub.status.idle":"2023-02-16T15:22:46.871763Z","shell.execute_reply.started":"2023-02-16T15:22:46.859760Z","shell.execute_reply":"2023-02-16T15:22:46.870916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nclass CVDataSet(Dataset):\n    def __init__(self, df, helmets, features, transforms,labels=None, data_type=None, is_scale_aug=True):\n        self.df = df[features].to_numpy()\n        self.game_key = df['game_key'].to_numpy()\n        self.game_play = df['game_play'].to_numpy()\n        self.play_id = df['play_id'].to_numpy()\n        self.frame = df['frame'].to_numpy()\n        self.nfl_player_id_x = df['nfl_player_id_x'].to_numpy()\n        self.nfl_player_id_y = df['nfl_player_id_y'].to_numpy()\n        if data_type in ['train', 'valid']:\n            self.contact_p1 = df['contact_shift1'].to_numpy()\n            self.contact_m1 = df['contact_shift-1'].to_numpy()\n            self.targets = df[['contact', 'contact_shift1', 'contact_shift-1']].to_numpy()\n\n        self.labels = labels\n        self.data_type = data_type\n        self.transforms = transforms\n        self.helmets = helmets\n        \n        self.is_scale_aug = is_scale_aug\n        \n        self.dummy_target = np.array([-1.0, -1.0, -1.0])\n        \n        r = np.zeros((Config.IMG_SIZE, Config.IMG_SIZE))\n        r[:] = 0.229\n        self.empty_img = np.moveaxis(np.stack([r]), 0, 2)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        if self.data_type in ['train', 'valid']:\n            target = self.targets[idx]#np.array([self.labels.iloc[idx]['contact'], self.contact_p1[idx], self.contact_m1[idx]])\n        else:\n            target = self.dummy_target\n            \n        feature = self.df[idx]\n        \n        \"\"\"endzone = create_img(self.frame[idx], self.game_play[idx] + '_Endzone.mp4', \n                             [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]], window=24, frame_interval=24, crop_size=256)\n        sideline = create_img(self.frame[idx], self.game_play[idx] + '_Sideline.mp4', \n                              [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]], window=24, frame_interval=24, crop_size=128)\"\"\"\n        endzone_type = -1\n        sideline_type = -1\n        \n        endzone = create_img_single_yolo(self.frame[idx], self.game_play[idx] + '_Endzone.mp4', \n                             [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]], window=24)\n        \n        try:\n            if endzone.sum().sum() == 0:\n                endzone_type = 0\n                endzone = create_img_single(self.frame[idx], self.game_play[idx] + '_Endzone.mp4', \n                                 [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]], window=24, frame_interval=1, crop_size=256)\n                if endzone.sum().sum() == 0:\n                    endzone_type = 1\n        except:\n            display(self.frame[idx], self.game_play[idx] + '_Endzone.mp4', [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]])\n            display(endzone.shape)\n            display(endzone)\n                \n            \n        \n        sideline = create_img_single_yolo(self.frame[idx], self.game_play[idx] + '_Sideline.mp4', \n                             [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]], window=24)\n        \n        if sideline.sum().sum() == 0:\n            sideline_type = 0\n            sideline = create_img_single(self.frame[idx], self.game_play[idx] + '_Sideline.mp4', \n                              [self.nfl_player_id_x[idx], self.nfl_player_id_y[idx]], window=24, frame_interval=1, crop_size=128)\n            if sideline.sum().sum() == 0:\n                sideline_type = 1\n            \n            \n        if self.data_type == 'train' and self.is_scale_aug:\n            try:\n                if random.random() < 0.5:\n                    scale = random.uniform(0.3, 0.8)\n                    sideline1 = cv2.resize(sideline[0], dsize=None, fx=scale, fy=scale)\n                    sideline2 = cv2.resize(sideline[1], dsize=None, fx=scale, fy=scale)\n                    sideline = np.array([sideline1, sideline2])\n            except:\n                pass\n                \n            try:\n                if random.random() < 0.5:\n                    scale = random.uniform(0.3, 0.8)\n                    endzone1 = cv2.resize(endzone[0], dsize=None, fx=scale, fy=scale)\n                    endzone2 = cv2.resize(endzone[1], dsize=None, fx=scale, fy=scale)\n                    endzone = np.array([endzone1, endzone2])\n            except:\n                pass\n            \n        sideline = np.moveaxis(sideline, 0, 2)\n        endzone = np.moveaxis(endzone, 0, 2)\n        \n        augmented = self.transforms(image=sideline)\n        sideline = augmented['image']\n        sideline = np.moveaxis(sideline, 2, 0)\n        \n        augmented = self.transforms(image=endzone)\n        endzone = augmented['image']\n        endzone = np.moveaxis(endzone, 2, 0)\n        \n        \n        feature = np.append(feature, endzone_type)\n        feature = np.append(feature, sideline_type)\n        \n        return feature, sideline, endzone, target","metadata":{"papermill":{"duration":0.035303,"end_time":"2021-09-27T02:55:09.041468","exception":false,"start_time":"2021-09-27T02:55:09.006165","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:46.873171Z","iopub.execute_input":"2023-02-16T15:22:46.873967Z","iopub.status.idle":"2023-02-16T15:22:46.896794Z","shell.execute_reply.started":"2023-02-16T15:22:46.873929Z","shell.execute_reply":"2023-02-16T15:22:46.895781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CVNet(nn.Module):\n    def __init__(self, num_features):\n        super(CVNet, self).__init__()\n\n        self.base_model = timm.create_model(Config.BACK_BONE, num_classes=0, pretrained=(not is_kaggle), in_chans=2)\n        base_model_features = self.base_model.num_features\n        \n        #self.sideline_base_model1 = timm.create_model(Config.BACK_BONE, num_classes=0, pretrained=False, in_chans=3)\n        #sideline_base_model1_features = self.sideline_base_model1.num_features\n        \n        #num_features_all = num_features + endzone_base_model1_features + sideline_base_model1_features\n        num_features_all = num_features + base_model_features * 2\n        \n        self.cls = nn.Sequential(\n            nn.Linear(num_features_all, int(num_features_all / 2)),\n            nn.ReLU(),\n            nn.Dropout(p=0.5),\n            nn.Linear(int(num_features_all / 2), int(num_features_all / 4)),\n            nn.ReLU(),\n            nn.Dropout(p=0.5),\n            nn.Linear(int(num_features_all / 4), int(num_features_all / 8)),\n            nn.ReLU(),\n            nn.Dropout(p=0.5),\n            nn.Linear(int(num_features_all / 8), Config.N_LABEL*3)\n        )\n        \n    def forward(self, x, sideline, endzone):\n\n        sideline_out = self.base_model(sideline)\n        endzone_out = self.base_model(endzone)\n        #sideline_out = self.sideline_base_model1(sideline)\n        \n        #x = torch.cat([endzone_out, sideline_out, x], 1)\n        #x = torch.cat([endzone_out, sideline_out], 1)\n        x = torch.cat([endzone_out, sideline_out, x], 1)\n        \n        x = self.cls(x)\n\n        return x","metadata":{"papermill":{"duration":0.034874,"end_time":"2021-09-27T02:55:09.102696","exception":false,"start_time":"2021-09-27T02:55:09.067822","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:46.898443Z","iopub.execute_input":"2023-02-16T15:22:46.898845Z","iopub.status.idle":"2023-02-16T15:22:46.910981Z","shell.execute_reply.started":"2023-02-16T15:22:46.898811Z","shell.execute_reply":"2023-02-16T15:22:46.910046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EarlyStopping:\n    def __init__(self, patience=7, verbose=False, delta=0, fold='', suffix=''):\n        self.patience = patience\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n        self.val_loss_min = np.Inf\n        self.delta = delta\n        self.suffix = suffix\n\n    def __call__(self, val_loss, model):\n\n        score = -val_loss\n\n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n        elif score < self.best_score + self.delta:\n            self.counter += 1\n            logger.info(f'EarlyStopping counter: {self.counter} out of {self.patience}')\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n            self.counter = 0\n\n    def save_checkpoint(self, val_loss, model):\n        '''Saves model when validation loss decrease.'''\n        if self.verbose:\n            logger.info(f'Validation loss decreased ({self.val_loss_min:.6f} --> {val_loss:.6f}).  Saving model ...')\n            \n        #if os.path.exists(CP_DIR / f'checkpoint_{NB}_{fold}.pt'):\n        #    shutil.move(CP_DIR / f'checkpoint_{NB}_{fold}.pt', CP_DIR / f'checkpoint_{NB}_{fold}-2.pt')\n        #torch.save(model.state_dict(), CP_DIR / f'checkpoint_{HOST}_{NB}_{fold}_{self.suffix}.pt')\n        torch.save(model.state_dict(), OUTPUT_DIR / f'{HOST}_{NB}_{fold}_checkpoint_{self.suffix}.pt')\n        self.val_loss_min = val_loss","metadata":{"papermill":{"duration":0.03708,"end_time":"2021-09-27T02:55:09.166253","exception":false,"start_time":"2021-09-27T02:55:09.129173","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:46.912407Z","iopub.execute_input":"2023-02-16T15:22:46.912985Z","iopub.status.idle":"2023-02-16T15:22:46.925020Z","shell.execute_reply.started":"2023-02-16T15:22:46.912950Z","shell.execute_reply":"2023-02-16T15:22:46.924045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\n\nclass SAM(torch.optim.Optimizer):\n    def __init__(self, params, base_optimizer, rho=0.05, adaptive=False, **kwargs):\n        assert rho >= 0.0, f\"Invalid rho, should be non-negative: {rho}\"\n\n        defaults = dict(rho=rho, adaptive=adaptive, **kwargs)\n        super(SAM, self).__init__(params, defaults)\n\n        self.base_optimizer = base_optimizer(self.param_groups, **kwargs)\n        self.param_groups = self.base_optimizer.param_groups\n        self.defaults.update(self.base_optimizer.defaults)\n\n    @torch.no_grad()\n    def first_step(self, zero_grad=False):\n        grad_norm = self._grad_norm()\n        for group in self.param_groups:\n            scale = group[\"rho\"] / (grad_norm + 1e-12)\n\n            for p in group[\"params\"]:\n                if p.grad is None: continue\n                self.state[p][\"old_p\"] = p.data.clone()\n                e_w = (torch.pow(p, 2) if group[\"adaptive\"] else 1.0) * p.grad * scale.to(p)\n                p.add_(e_w)  # climb to the local maximum \"w + e(w)\"\n\n        if zero_grad: self.zero_grad()\n\n    @torch.no_grad()\n    def second_step(self, zero_grad=False):\n        for group in self.param_groups:\n            for p in group[\"params\"]:\n                if p.grad is None: continue\n                p.data = self.state[p][\"old_p\"]  # get back to \"w\" from \"w + e(w)\"\n\n        self.base_optimizer.step()  # do the actual \"sharpness-aware\" update\n\n        if zero_grad: self.zero_grad()\n\n    @torch.no_grad()\n    def step(self, closure=None):\n        assert closure is not None, \"Sharpness Aware Minimization requires closure, but it was not provided\"\n        closure = torch.enable_grad()(closure)  # the closure should do a full forward-backward pass\n\n        self.first_step(zero_grad=True)\n        closure()\n        self.second_step()\n\n    def _grad_norm(self):\n        shared_device = self.param_groups[0][\"params\"][0].device  # put everything on the same device, in case of model parallelism\n        norm = torch.norm(\n                    torch.stack([\n                        ((torch.abs(p) if group[\"adaptive\"] else 1.0) * p.grad).norm(p=2).to(shared_device)\n                        for group in self.param_groups for p in group[\"params\"]\n                        if p.grad is not None\n                    ]),\n                    p=2\n               )\n        return norm\n\n    def load_state_dict(self, state_dict):\n        super().load_state_dict(state_dict)\n        self.base_optimizer.param_groups = self.param_groups","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.926491Z","iopub.execute_input":"2023-02-16T15:22:46.927093Z","iopub.status.idle":"2023-02-16T15:22:46.942860Z","shell.execute_reply.started":"2023-02-16T15:22:46.927057Z","shell.execute_reply":"2023-02-16T15:22:46.941737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ndevice","metadata":{"papermill":{"duration":0.084495,"end_time":"2021-09-27T02:55:09.277010","exception":false,"start_time":"2021-09-27T02:55:09.192515","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:46.944186Z","iopub.execute_input":"2023-02-16T15:22:46.945102Z","iopub.status.idle":"2023-02-16T15:22:46.956220Z","shell.execute_reply.started":"2023-02-16T15:22:46.945067Z","shell.execute_reply":"2023-02-16T15:22:46.955199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfeature, sideline, endzone, target = CVDataSet(test_df, test_helmets, features, get_test_augmentation(), labels=sample_submission_df, data_type='test')[0]\nplt.imshow(sideline[0,:,:])\nplt.show()\n\nplt.imshow(endzone[0,:,:])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:46.958009Z","iopub.execute_input":"2023-02-16T15:22:46.958356Z","iopub.status.idle":"2023-02-16T15:22:47.479316Z","shell.execute_reply.started":"2023-02-16T15:22:46.958323Z","shell.execute_reply":"2023-02-16T15:22:47.478488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"fold_game_plays = []\n\nskf = StratifiedGroupKFold(n_splits=Config.N_FOLD, random_state=Config.RANDOM_SATE, shuffle=True)\nfor fold, (train_index, test_index) in enumerate(skf.split(train_df, train_df['contact'], train_df['game_play'])):\n    fold_game_plays.append(train_labels_df.iloc[train_index]['game_play'].unique())\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:22:47.480665Z","iopub.execute_input":"2023-02-16T15:22:47.481602Z","iopub.status.idle":"2023-02-16T15:22:47.487752Z","shell.execute_reply.started":"2023-02-16T15:22:47.481568Z","shell.execute_reply":"2023-02-16T15:22:47.486802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    skf = StratifiedGroupKFold(n_splits=Config.N_FOLD, random_state=Config.RANDOM_SATE, shuffle=True)\n    for fold, (train_index, test_index) in enumerate(skf.split(train_df, train_df['contact'], train_df['game_play'])):\n        print(f'====== {fold} ======')\n\n        net = CVNet(len(features) + 2)\n        net.to(device)\n\n        #criterion = nn.MSELoss()\n        criterion = nn.BCEWithLogitsLoss()\n        base_optimizer = optim.AdamW\n        optimizer = SAM(net.parameters(), base_optimizer, lr=Config.LR, weight_decay=1.0e-02)\n\n        train_labels = train_labels_df.iloc[train_index].reset_index(drop=True)\n        valid_labels = train_labels_df.iloc[test_index].reset_index(drop=True)\n\n        train_game_play_df = train_df[train_df['game_play'].isin(train_labels['game_play'].unique())].reset_index(drop=True)\n        valid_game_play_df = train_df[train_df['game_play'].isin(valid_labels['game_play'].unique())].reset_index(drop=True)\n\n\n        train_dataset = CVDataSet(train_game_play_df, train_helmets, features, get_augmentation(), labels=train_labels, data_type='train')\n        valid_dataset = CVDataSet(valid_game_play_df, train_helmets, features, get_test_augmentation(), labels=valid_labels, data_type='valid')\n\n        trainloader = DataLoader(train_dataset, batch_size=Config.BATCH_SIZE, shuffle=True, drop_last=True, pin_memory=True, num_workers=Config.NUM_WORKERS)\n        validloader = DataLoader(valid_dataset, batch_size=Config.BATCH_SIZE , pin_memory=True, num_workers=Config.NUM_WORKERS)\n        \n        soft_train_dataset = CVDataSet(train_game_play_df, train_helmets, features, get_soft_augmentation(), labels=train_labels, data_type='train', is_scale_aug=False)\n        soft_trainloader = DataLoader(soft_train_dataset, batch_size=Config.BATCH_SIZE, shuffle=True, drop_last=True, pin_memory=True, num_workers=Config.NUM_WORKERS)\n\n        early_stopping = EarlyStopping(patience=Config.PATIENCE, verbose=True, fold=fold, suffix='pair')\n        scheduler = torch.optim.lr_scheduler.OneCycleLR(optimizer, epochs=Config.EPOCH, steps_per_epoch=len(trainloader), max_lr=Config.MAX_LR, pct_start= 0.1, anneal_strategy='cos', div_factor= 1.0e+3, final_div_factor= 1.0e+4)\n        # scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=Config.T_MAX, eta_min=Config.ETA_MIN)\n        # scheduler = CosineAnnealingWarmupRestarts(optimizer, first_cycle_steps=int((len(trainloader) * Config.EPOCH) / 5), cycle_mult=1.0, max_lr=Config.LR, min_lr=Config.ETA_MIN, warmup_steps=int((len(trainloader) * Config.EPOCH) / 50), gamma=Config.SCHEDULER_GAMMA)\n\n        val_metrics = []\n        val_loss = []\n        learning_rates = []\n\n        for name, param in net.named_parameters():\n            if not name.startswith('base_model'):\n                param.requires_grad = False #固定\n            else:\n                param.requires_grad = True\n\n        for epoch in range(Config.EPOCH):\n\n\n            if epoch == 1:\n                for name, param in net.named_parameters():\n                    param.requires_grad = True\n\n            running_loss = 0.0\n            train_rmse_list = []\n            \n            if epoch <= 9: \n                loader = trainloader\n            else:\n                loader = soft_trainloader\n            \n            n_iter = len(loader)\n            with tqdm(enumerate(loader), total=n_iter, smoothing=0) as pbar:\n                for i, (feature, sideline, endzone, target) in pbar:\n\n                    net.train()\n                    # zero the parameter gradients\n                    #optimizer.zero_grad()\n\n                    feature, sideline, endzone, target = feature.to(device).float(), sideline.to(device).float(), endzone.to(device).float(), target.to(device).float()\n\n                    outputs = net(feature, sideline, endzone)\n\n                    loss = criterion(outputs.squeeze(), target)\n\n                    loss.backward()                                 # Backward pass\n                    optimizer.first_step(zero_grad=True)     \n\n                    outputs = net(feature, sideline, endzone).squeeze()\n                    criterion(outputs, target).backward()\n\n                    optimizer.second_step(zero_grad=True)\n                    net.zero_grad()     \n\n\n                    # print statistics\n                    running_loss += loss.item()\n\n                    outputs_np = outputs.to('cpu').detach().numpy().copy()\n\n                    pbar.set_postfix(OrderedDict(\n                        epoch=\"{:>10}\".format(epoch), loss=\"{:.4f}\".format(loss.item())\n                    ))\n                    scheduler.step()\n\n\n            val_preds = []\n            losses = []\n            n_iter_val = len(validloader)\n            for j, (feature, sideline, endzone, target) in tqdm(enumerate(validloader), total=n_iter_val, smoothing=0):\n                net.eval()\n\n                with torch.no_grad():\n                    feature, sideline, endzone, target = feature.to(device).float(), sideline.to(device).float(), endzone.to(device).float(), target.to(device).float()\n                    outputs = net(feature, sideline, endzone)\n                    loss = criterion(outputs.squeeze(), target)\n                    outputs = outputs.sigmoid()\n                    outputs_np = outputs.to('cpu').detach().numpy().copy()[:, 0]\n                    val_preds.append(outputs_np)\n                    losses.append(loss.item())\n\n            auc = roc_auc_score(valid_game_play_df['contact'].to_numpy(), np.hstack(val_preds))\n            logger.info('auc:{:.4f}, loss:{:.4f}'.format(auc, np.mean(losses)))\n\n            lr = optimizer.param_groups[0]['lr']\n\n            val_metrics.append(auc)\n            val_loss.append(np.mean(losses))\n            learning_rates.append(lr)\n\n            early_stopping(-auc, net)\n\n            if early_stopping.early_stop:\n                logger.info(\"Early stopping\")\n                net.load_state_dict(torch.load(CP_DIR / f'checkpoint_{NB}_{fold}.pt'))\n                #cv_scores[f'cv{fold}'] = early_stopping.best_score\n                break\n\n        fig = plt.figure()\n        ax1 = fig.add_subplot(111)\n        ax1.plot(learning_rates)\n        ax2 = ax1.twinx()\n        ax2.plot(val_metrics)\n        plt.show()\n\n        del net, validloader, trainloader, train_dataset, valid_dataset\n        torch.cuda.empty_cache()\n        gc.collect()\n","metadata":{"papermill":{"duration":0.049482,"end_time":"2021-09-27T02:55:09.353427","exception":false,"start_time":"2021-09-27T02:55:09.303945","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:47.489506Z","iopub.execute_input":"2023-02-16T15:22:47.490258Z","iopub.status.idle":"2023-02-16T15:22:47.517343Z","shell.execute_reply.started":"2023-02-16T15:22:47.490215Z","shell.execute_reply":"2023-02-16T15:22:47.516260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## validation","metadata":{}},{"cell_type":"code","source":"if not is_kaggle:\n    cv_scores = {}\n    oof = np.zeros((len(train_df), Config.N_LABEL))\n    oof_all = np.zeros((len(train_df), Config.N_LABEL * 3))\n    skf = StratifiedGroupKFold(n_splits=Config.N_FOLD, random_state=Config.RANDOM_SATE, shuffle=True)\n    for fold, (train_index, test_index) in enumerate(skf.split(train_labels_df, train_labels_df['contact'], train_labels_df['game_play'])):\n        print(test_index)\n        print(f'====== {fold} ======')\n\n        net = CVNet(len(features) + 2)\n        net.to(device)\n\n        net.load_state_dict(torch.load(OUTPUT_DIR / f'{HOST}_{NB}_{fold}_checkpoint_pair.pt'))\n        net.eval()\n\n        valid_labels = train_labels_df.iloc[test_index].reset_index(drop=True)\n        valid_game_play_df = train_df[train_df['game_play'].isin(valid_labels['game_play'].unique())].reset_index(drop=True)\n\n        valid_dataset = CVDataSet(valid_game_play_df, train_helmets, features, get_test_augmentation(), labels=valid_labels, data_type='valid')\n        validloader = DataLoader(valid_dataset, batch_size=Config.BATCH_SIZE * 2, num_workers=Config.NUM_WORKERS, pin_memory=True)\n\n        preds = []\n        preds_all = []\n        for i, (feature, sideline, endzone, target) in tqdm(enumerate(validloader), total=len(validloader), smoothing=0):\n\n            with torch.no_grad():\n\n                feature, sideline, endzone, target = feature.to(device).float(), sideline.to(device).float(), endzone.to(device).float(), target.to(device).float()\n                outputs = net(feature, sideline, endzone)\n                outputs_np = outputs.sigmoid().to('cpu').detach().numpy().copy()\n                preds.append(outputs_np[:, 0])\n                preds_all.append(outputs_np)\n\n        auc = roc_auc_score(valid_game_play_df['contact'].to_numpy(), np.hstack(preds))\n        auc_all = roc_auc_score(valid_game_play_df[['contact', 'contact_shift1', 'contact_shift-1']].to_numpy(), np.vstack(preds_all))\n        print(auc, auc_all)\n\n        oof[test_index] = np.hstack(preds).reshape(-1, 1)\n        oof_all[test_index, :] = np.vstack(preds_all)\n        cv_scores[f'cv{fold}'] = auc\n\n        del net, valid_dataset, validloader\n        torch.cuda.empty_cache()\n        gc.collect()\n\n    print(np.mean([v for k, v in cv_scores.items()]))","metadata":{"papermill":{"duration":0.040168,"end_time":"2021-09-27T02:55:09.420226","exception":false,"start_time":"2021-09-27T02:55:09.380058","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:47.519115Z","iopub.execute_input":"2023-02-16T15:22:47.519510Z","iopub.status.idle":"2023-02-16T15:22:47.535599Z","shell.execute_reply.started":"2023-02-16T15:22:47.519475Z","shell.execute_reply":"2023-02-16T15:22:47.534657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.034584,"end_time":"2021-09-27T02:55:09.481611","exception":false,"start_time":"2021-09-27T02:55:09.447027","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_preds = np.zeros(len(test_df))\ny_preds_all = np.zeros((len(test_df), 3))\n\ntest_dataset = CVDataSet(test_df, test_helmets, features, get_test_augmentation(), labels=sample_submission_df, data_type='test')\ntestloader = DataLoader(test_dataset, batch_size=Config.BATCH_SIZE * 2, pin_memory=True, num_workers=Config.NUM_WORKERS)\n\nfor fold in range(Config.N_FOLD):\n\n    net = CVNet(len(features)+2)\n    net.to(device)\n    if not is_kaggle:\n        net.load_state_dict(torch.load(OUTPUT_DIR / f'{HOST}_{NB}_{fold}_checkpoint_pair.pt'))\n    else:\n        net.load_state_dict(torch.load(f'/kaggle/input/nfl-models/{HOST}_{NB}_{fold}_checkpoint_pair.pt'))\n\n    #fold_preds = []\n    fold_preds_all = []\n    for i, (feature, sideline, endzone, target) in tqdm(enumerate(testloader), total=len(testloader), smoothing=0):\n        net.eval()\n\n        with torch.no_grad():\n            feature, sideline, endzone, target = feature.to(device).float(), sideline.to(device).float(), endzone.to(device).float(), target.to(device).float()\n            outputs = net(feature, sideline, endzone)\n            outputs_np = outputs.sigmoid().to('cpu').detach().numpy().copy()\n            \n            #fold_preds.append(outputs_np[:, 0])\n            fold_preds_all.append(outputs_np)\n    \n    #y_preds += np.hstack(fold_preds).reshape(-1) / Config.N_FOLD\n    y_preds_all += np.vstack(fold_preds_all) / Config.N_FOLD","metadata":{"papermill":{"duration":0.036379,"end_time":"2021-09-27T02:55:09.544913","exception":false,"start_time":"2021-09-27T02:55:09.508534","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-02-16T15:22:47.538917Z","iopub.execute_input":"2023-02-16T15:22:47.539521Z","iopub.status.idle":"2023-02-16T15:28:19.504650Z","shell.execute_reply.started":"2023-02-16T15:22:47.539492Z","shell.execute_reply":"2023-02-16T15:28:19.502306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del net, testloader, test_dataset, outputs, outputs_np, feature, sideline, endzone, target, fold_preds_all\ndel data_dict\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:19.508185Z","iopub.execute_input":"2023-02-16T15:28:19.509268Z","iopub.status.idle":"2023-02-16T15:28:19.833716Z","shell.execute_reply.started":"2023-02-16T15:28:19.509215Z","shell.execute_reply":"2023-02-16T15:28:19.832704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df[['contact_id']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    train_df['pred'] = oof_all[:, 0]\n    train_df['pred_s1'] = oof_all[:, 1]\n    train_df['pred_s-1'] = oof_all[:, 2]\n    train_hight_distance_df['pred'] = 0\n    train_pred_df = pd.concat([train_df[['contact', 'contact_id', 'pred', 'pred_s1', 'pred_s-1']], train_hight_distance_df[['contact', 'contact_id', 'pred']]], axis=0).reset_index(drop=True)\n    train_pred_df = train_pred_df[['contact', 'contact_id', 'pred', 'pred_s1', 'pred_s-1']].copy()\n    \n    train_pred_df['game_play'] = train_pred_df['contact_id'].apply(lambda x : x.split('_')[0] + '_' + x.split('_')[1])\n    train_pred_df['step'] = train_pred_df['contact_id'].apply(lambda x : int(x.split('_')[2]))\n    train_pred_df['nfl_player_pair'] = train_pred_df['contact_id'].apply(lambda x : x.split('_')[3] + '_' + x.split('_')[4])\n\ntest_df['pred'] = y_preds_all[:, 0]\ntest_df['pred_s1'] = y_preds_all[:, 1]\ntest_df['pred_s-1'] = y_preds_all[:, 2]\n\n\ndel y_preds_all\ngc.collect()\n\ntest_hight_distance_df['pred'] = 0\ntest_pred_df = pd.concat([test_df[['contact_id', 'pred', 'pred_s1', 'pred_s-1']], test_hight_distance_df[['contact_id', 'pred']]], axis=0).reset_index(drop=True)\ntest_pred_df = test_pred_df[['contact_id', 'pred', 'pred_s1', 'pred_s-1']].copy()\n\ntest_pred_df['game_play'] = test_pred_df['contact_id'].apply(lambda x : x.split('_')[0] + '_' + x.split('_')[1])\ntest_pred_df['step'] = test_pred_df['contact_id'].apply(lambda x : int(x.split('_')[2]))\ntest_pred_df['nfl_player_pair'] = test_pred_df['contact_id'].apply(lambda x : x.split('_')[3] + '_' + x.split('_')[4])","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:19.835152Z","iopub.execute_input":"2023-02-16T15:28:19.836595Z","iopub.status.idle":"2023-02-16T15:28:20.201317Z","shell.execute_reply.started":"2023-02-16T15:28:19.836537Z","shell.execute_reply":"2023-02-16T15:28:20.200269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df, test_hight_distance_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"%%time\n\ndef opt_thr_2(window_pair, window_g, n_trials=100, log_level=optuna.logging.WARNING):\n    train_pred_ma_pair_df['pred_ma'] = train_pred_ma_pair_df.groupby(['game_play', 'nfl_player_pair'])['pred'].rolling(window_pair, center=True, min_periods=1).mean().to_frame('pred_ma').reset_index()['pred_ma']\n    pair_np = train_pred_ma_pair_df['pred_ma'].to_numpy()\n    \n    train_pred_ma_g_df['pred_ma'] = train_pred_ma_g_df.groupby(['game_play', 'nfl_player_pair'])['pred'].rolling(window_g, center=True, min_periods=1).mean().to_frame('pred_ma').reset_index()['pred_ma']\n    g_np= train_pred_ma_g_df['pred_ma'].to_numpy()\n    \n    t = np.hstack([train_pred_ma_pair_df['contact'].to_numpy(), train_pred_ma_g_df['contact'].to_numpy()])\n    \n    def calc_score_pair(thr_pair, thr_g):\n        \n        pair_contact_pred = np.where(pair_np > thr_pair, 1, 0)\n        g_contact_pred = np.where(g_np > thr_g, 1, 0)\n\n        score = matthews_corrcoef(t, np.hstack([pair_contact_pred, g_contact_pred]))\n\n        return score\n\n    def objective_pair(trial):\n\n        thr_pair = trial.suggest_uniform('thr_pair', 1e-10, 5e-1)\n        thr_g = trial.suggest_uniform('thr_g', 1e-10, 5e-1)\n\n        score = calc_score_pair(thr_pair, thr_g)\n        return score\n\n    optuna.logging.set_verbosity(log_level)\n\n    pair_study = optuna.create_study(direction='maximize')\n\n    pair_study.optimize(objective_pair, n_trials=n_trials, n_jobs=10)\n    \n    return pair_study\n\nif not is_kaggle:\n    train_pred_ma_pair_df = train_pred_df[~(train_pred_df['nfl_player_pair'].apply(lambda x : x.endswith('G')))].reset_index(drop=True)\n    train_pred_ma_g_df = train_pred_df[train_pred_df['nfl_player_pair'].apply(lambda x : x.endswith('G'))].reset_index(drop=True)\n    \n    train_pred_ma_pair_df = train_pred_ma_pair_df.sort_values(['game_play', 'nfl_player_pair', 'step']).reset_index(drop=True)\n    train_pred_ma_g_df = train_pred_ma_g_df.sort_values(['game_play', 'nfl_player_pair', 'step']).reset_index(drop=True)\n    \n    opt_result = []\n\n    for window_pair in range(1, 10):\n        for window_g in range(1, 20):\n            print(f'{window_pair}, {window_g}')\n            study = opt_thr_2(window_pair, window_g,n_trials=100)\n            print(study.best_params['thr_pair'], study.best_params['thr_g'], study.best_value)\n            opt_result.append({\n                'window_pair':window_pair,\n                'window_g':window_g,\n                'thr_pair':study.best_params['thr_pair'],\n                'thr_g':study.best_params['thr_g'],\n                'score':study.best_value\n            })\n\n    opt_result_df = pd.DataFrame(opt_result)\n    display(opt_result_df.sort_values('score'))\n    print(opt_result_df.sort_values('score').iloc[-1]['thr_pair'])\n    print(opt_result_df.sort_values('score').iloc[-1]['thr_g'])\n    \n    window_pair_value = opt_result_df.sort_values('score').iloc[-1]['window_pair']\n    window_g_value = opt_result_df.sort_values('score').iloc[-1]['window_g']\n    thr_pair_value = opt_result_df.sort_values('score').iloc[-1]['thr_pair']\n    thr_g_value = opt_result_df.sort_values('score').iloc[-1]['thr_g']\n    \n    opt_params = {'window_pair':window_pair_value, 'window_g':window_g_value, 'thr_pair':thr_pair_value, 'thr_g':thr_g_value}\n    \n    to_pickle(f'../output/{HOST}_{NB}_opt_param.pkl', opt_params)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.202943Z","iopub.execute_input":"2023-02-16T15:28:20.203337Z","iopub.status.idle":"2023-02-16T15:28:20.216894Z","shell.execute_reply.started":"2023-02-16T15:28:20.203297Z","shell.execute_reply":"2023-02-16T15:28:20.215556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    to_pickle(f'../output/{HOST}_{NB}_train_pred_df.pkl', train_pred_df)\n    to_pickle(f'../output/{HOST}_{NB}_test_pred_df.pkl', test_pred_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.219310Z","iopub.execute_input":"2023-02-16T15:28:20.219638Z","iopub.status.idle":"2023-02-16T15:28:20.228831Z","shell.execute_reply.started":"2023-02-16T15:28:20.219600Z","shell.execute_reply":"2023-02-16T15:28:20.227662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"#atodekesu\ntrain_pred_df = unpickle(f'../output/{HOST}_exp0135_cnn_multitask_0126base_train_pred_df.pkl')\ntest_pred_df = unpickle(f'../output/{HOST}_exp0135_cnn_multitask_0126base_test_pred_df.pkl')\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.231658Z","iopub.execute_input":"2023-02-16T15:28:20.232365Z","iopub.status.idle":"2023-02-16T15:28:20.241532Z","shell.execute_reply.started":"2023-02-16T15:28:20.232325Z","shell.execute_reply":"2023-02-16T15:28:20.239846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## testの予測値相関確認(pp前)","metadata":{}},{"cell_type":"code","source":"if is_kaggle:\n    _test_df = unpickle(f'/kaggle/input/nfl-models/{HOST}_{NB}_test_pred_df.pkl')\n    if len(test_pred_df) == len(_test_df):#subのときは動かないように\n        plt.figure(figsize=(10, 10))\n        plt.scatter(test_pred_df['pred'], _test_df['pred'], s=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.243039Z","iopub.execute_input":"2023-02-16T15:28:20.243370Z","iopub.status.idle":"2023-02-16T15:28:20.617515Z","shell.execute_reply.started":"2023-02-16T15:28:20.243341Z","shell.execute_reply":"2023-02-16T15:28:20.616531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = 'contact'\nfeatures = ['pred', 'player_distance', 'speed_x', 'speed_y', 'G_flag', 'step', 'same_team', 'pred_s1', 'pred_s-1']","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.619053Z","iopub.execute_input":"2023-02-16T15:28:20.619943Z","iopub.status.idle":"2023-02-16T15:28:20.625285Z","shell.execute_reply.started":"2023-02-16T15:28:20.619901Z","shell.execute_reply":"2023-02-16T15:28:20.624301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_pred_df = train_pred_df.sort_values(['game_play', 'nfl_player_pair', 'step']).reset_index(drop=True)\ntest_pred_df = test_pred_df.sort_values(['game_play', 'nfl_player_pair', 'step']).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.626606Z","iopub.execute_input":"2023-02-16T15:28:20.627407Z","iopub.status.idle":"2023-02-16T15:28:20.666058Z","shell.execute_reply.started":"2023-02-16T15:28:20.627371Z","shell.execute_reply":"2023-02-16T15:28:20.665123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## pp LGBM用の特徴量作成","metadata":{}},{"cell_type":"code","source":"if not is_kaggle:\n    \n    \n    train_pred_df[f'pred_s1_s-1'] = train_pred_df.groupby(['game_play', 'nfl_player_pair'])['pred_s1'].shift(-1)\n    train_pred_df[f'pred_s-1_s1'] = train_pred_df.groupby(['game_play', 'nfl_player_pair'])['pred_s-1'].shift(1)\n    features.append(f'pred_s1_s-1')\n    features.append(f'pred_s-1_s1')\n    \n    for shift in range(-13, 14):\n        train_pred_df[f'pred_shift{shift}'] = train_pred_df.groupby(['game_play', 'nfl_player_pair'])['pred'].shift(shift)\n        features.append(f'pred_shift{shift}')\n\n    train_pred_df = train_pred_df.merge(pp_feature_train[['contact_id', 'speed_x', 'speed_y', 'player_distance', 'G_flag', 'same_team']], how='left', on='contact_id')\n    \n    for shift in range(-12, 13):\n        train_pred_df[f'player_distance_shift{shift}'] = train_pred_df.groupby(['game_play', 'nfl_player_pair'])['player_distance'].shift(shift)\n        features.append(f'player_distance_shift{shift}')\n        \n    for shift in range(-2, 3):\n        train_pred_df[f'speed_x_shift{shift}'] = train_pred_df.groupby(['game_play', 'nfl_player_pair'])['speed_x'].shift(shift)\n        features.append(f'speed_x_shift{shift}')\n    \n    train_pred_df = train_pred_df.sort_values(['game_play', 'nfl_player_pair', 'step']).reset_index(drop=True)\n    \n# test\n\ntest_pred_df[f'pred_s1_s-1'] = test_pred_df.groupby(['game_play', 'nfl_player_pair'])['pred_s1'].shift(-1)\ntest_pred_df[f'pred_s-1_s1'] = test_pred_df.groupby(['game_play', 'nfl_player_pair'])['pred_s-1'].shift(1)\nfeatures.append(f'pred_s1_s-1')\nfeatures.append(f'pred_s-1_s1')\n\nfor shift in range(-13, 14):\n    test_pred_df[f'pred_shift{shift}'] = test_pred_df.groupby(['game_play', 'nfl_player_pair'])['pred'].shift(shift)\n    features.append(f'pred_shift{shift}')\n\ntest_pred_df = test_pred_df.merge(pp_feature_test[['contact_id', 'speed_x', 'speed_y', 'player_distance', 'G_flag', 'same_team']], how='left', on='contact_id')\n\nfor shift in range(-12, 13):\n    test_pred_df[f'player_distance_shift{shift}'] = test_pred_df.groupby(['game_play', 'nfl_player_pair'])['player_distance'].shift(shift)\n    features.append(f'player_distance_shift{shift}')\n\nfor shift in range(-2, 3):\n    test_pred_df[f'speed_x_shift{shift}'] = test_pred_df.groupby(['game_play', 'nfl_player_pair'])['speed_x'].shift(shift)\n    features.append(f'speed_x_shift{shift}')\n    \ntest_pred_df = test_pred_df.sort_values(['game_play', 'nfl_player_pair', 'step']).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:20.667630Z","iopub.execute_input":"2023-02-16T15:28:20.667968Z","iopub.status.idle":"2023-02-16T15:28:21.397932Z","shell.execute_reply.started":"2023-02-16T15:28:20.667935Z","shell.execute_reply":"2023-02-16T15:28:21.396914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_pred_df['same_team'] = train_pred_df['same_team'].astype(np.float)\ntest_pred_df['same_team'] = test_pred_df['same_team'].astype(np.float)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.399587Z","iopub.execute_input":"2023-02-16T15:28:21.399958Z","iopub.status.idle":"2023-02-16T15:28:21.408562Z","shell.execute_reply.started":"2023-02-16T15:28:21.399921Z","shell.execute_reply":"2023-02-16T15:28:21.407586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"target = 'contact'\nfeatures = ['pred', 'pred_shift1', 'pred_shift2', 'pred_shift3', 'pred_shift4', 'pred_shift5', 'pred_shift6', 'pred_shift-1', 'pred_shift-2', 'pred_shift-3', 'pred_shift-4', 'pred_shift-5', 'pred_shift-6', 'player_distance', 'speed_x', 'speed_y', 'G_flag', 'step', 'player_distance_shift1', 'player_distance_shift2', 'player_distance_shift3', 'player_distance_shift4', 'player_distance_shift5', 'player_distance_shift6', 'player_distance_shift-1', 'player_distance_shift-2', 'player_distance_shift-3', 'player_distance_shift-4', 'player_distance_shift-5', 'player_distance_shift-6', 'same_team']\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.410269Z","iopub.execute_input":"2023-02-16T15:28:21.410902Z","iopub.status.idle":"2023-02-16T15:28:21.421022Z","shell.execute_reply.started":"2023-02-16T15:28:21.410865Z","shell.execute_reply":"2023-02-16T15:28:21.419532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LightGBM学習","metadata":{}},{"cell_type":"code","source":"if not is_kaggle:\n    #import lightgbm as lgb\n    import optuna.integration.lightgbm as lgb\n\n    def evaluation(true, pred):\n        return roc_auc_score(true, pred)\n\n    params = {\n        'n_estimators':5000,\n        'boosting_type': 'gbdt',\n        'metric': 'binary_logloss',\n        'objective': 'binary',\n        'n_jobs': -1,\n        'seed': 47,\n        'learning_rate': 0.01,\n    }\n\n    oof_preds = np.zeros(len(train_pred_df))\n    lgbm_models = []\n    cv_scores = {}\n    skf = StratifiedGroupKFold(n_splits=3, random_state=47, shuffle=True)\n    for fold, (train_index, test_index) in enumerate(skf.split(train_pred_df, train_pred_df['contact'], train_pred_df['game_play'])):\n\n        print(f'====== fold {fold} ======')\n\n        # TrainとTestに分割\n        x_train, x_val = train_pred_df.copy().iloc[train_index][features], train_pred_df.copy().iloc[test_index][features]\n        y_train, y_val =  train_pred_df.iloc[train_index][target], train_pred_df.iloc[test_index][target]\n\n        train_features = x_train.columns.to_list()\n\n        # create Dataset\n        train_set = lgb.Dataset(x_train, y_train, free_raw_data=False)\n        val_set = lgb.Dataset(x_val, y_val, free_raw_data=False)\n\n        # train\n        model = lgb.train(params, train_set, valid_sets=[train_set, val_set], verbose_eval=100, early_stopping_rounds=100)#, feval=rmsle_eval)\n\n        lgbm_models.append(model)\n\n        fold_pred = model.predict(x_val)\n\n        score = evaluation(y_val, fold_pred)\n        cv_scores[f'cv{fold}'] = score\n\n        oof_preds[test_index] = fold_pred\n\n        print(f'cv score is {score}')","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.422476Z","iopub.execute_input":"2023-02-16T15:28:21.423290Z","iopub.status.idle":"2023-02-16T15:28:21.434531Z","shell.execute_reply.started":"2023-02-16T15:28:21.423255Z","shell.execute_reply":"2023-02-16T15:28:21.433570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"1","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.437799Z","iopub.execute_input":"2023-02-16T15:28:21.438143Z","iopub.status.idle":"2023-02-16T15:28:21.448696Z","shell.execute_reply.started":"2023-02-16T15:28:21.438116Z","shell.execute_reply":"2023-02-16T15:28:21.447603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"if not is_kaggle:\n    import lightgbm as lgb\n\n    def evaluation(true, pred):\n        return roc_auc_score(true, pred)S\n\n    from catboost import CatBoostClassifier, CatBoostRegressor, Pool\n\n    params = {\n        'learning_rate': 0.01, \n        'loss_function': 'Logloss',\n        'verbose':100,\n        'iterations':10000,\n        'early_stopping_rounds':100,\n        'task_type': 'GPU',\n    }\n\n    oof_preds = np.zeros(len(train_pred_df))\n    lgbm_models = []\n    cv_scores = {}\n    skf = StratifiedGroupKFold(n_splits=3, random_state=47, shuffle=True)\n    for fold, (train_index, test_index) in enumerate(skf.split(train_pred_df, train_pred_df['contact'], train_pred_df['game_play'])):\n\n        print(f'====== fold {fold} ======')\n\n        # TrainとTestに分割\n        #x_train, x_val = train_pred_df.copy().iloc[train_index][features], train_pred_df.copy().iloc[test_index][features]\n        #y_train, y_val = train_pred_df.iloc[train_index][target], train_pred_df.iloc[test_index][target]\n        \n        x_train, x_val = train_pred_df[train_pred_df['game_play'].isin(fold_game_plays[fold])][features], train_pred_df[~(train_pred_df['game_play'].isin(fold_game_plays[fold]))][features]\n        y_train, y_val = train_pred_df[train_pred_df['game_play'].isin(fold_game_plays[fold])][target], train_pred_df[~(train_pred_df['game_play'].isin(fold_game_plays[fold]))][target]\n\n        test_index = train_pred_df[~(train_pred_df['game_play'].isin(fold_game_plays[fold]))].index.tolist()\n\n        # create Dataset\n        train_pool = Pool(x_train, y_train)\n        valid_pool = Pool(x_val, y_val)\n\n        # train\n        model = CatBoostClassifier(**params)\n        model.fit(train_pool, eval_set=valid_pool)\n        lgbm_models.append(model)\n\n        fold_pred = model.predict(valid_pool, prediction_type='Probability')[:, 1]\n\n        score = evaluation(y_val, fold_pred)\n        cv_scores[f'cv{fold}'] = score\n\n        oof_preds[test_index] = fold_pred\n\n        print(f'cv score is {score}')\n\n    oof_score = evaluation(train_pred_df[target], oof_preds)\n    print(f'OOF score is {oof_score}')\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.450362Z","iopub.execute_input":"2023-02-16T15:28:21.451279Z","iopub.status.idle":"2023-02-16T15:28:21.460072Z","shell.execute_reply.started":"2023-02-16T15:28:21.451240Z","shell.execute_reply":"2023-02-16T15:28:21.458863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"if not is_kaggle:\n    df_importance = None\n\n    for i, model in enumerate(lgbm_models):\n        if df_importance is None:\n            _df = pd.DataFrame([model.feature_importance(importance_type='gain'), train_features]).T\n            _df.columns = [f'model_{i}_gain', 'feature']\n            df_importance = _df\n        else:\n            _df = pd.DataFrame([model.feature_importance(importance_type='gain'), train_features]).T\n            _df.columns = [f'model_{i}_gain', 'feature']\n            df_importance = df_importance.merge(_df, how='outer', on='feature')\n\n    df_imp = df_importance\n    df_imp['mean'] = df_imp[[f'model_{i}_gain' for i in range(len(lgbm_models))]].mean(axis=1)\n    order = df_imp.sort_values('mean', ascending=False)['feature'].tolist()\n\n    df_imp = pd.melt(df_imp, id_vars=['feature'], value_vars=[f'model_{i}_gain' for i in range(len(lgbm_models))])\n    df_imp['value'] = df_imp['value'].astype(float)\n\n    fig, ax = plt.subplots(figsize=(len(df_imp['feature'].drop_duplicates()) * .4, 5))\n    sns.boxenplot(x=\"feature\", y=\"value\", data=df_imp, order=order)\n    ax.tick_params(axis='x', rotation=90)\n    ax.set_title('feature importance')\n    plt.show()\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.461601Z","iopub.execute_input":"2023-02-16T15:28:21.462361Z","iopub.status.idle":"2023-02-16T15:28:21.473362Z","shell.execute_reply.started":"2023-02-16T15:28:21.462325Z","shell.execute_reply":"2023-02-16T15:28:21.472677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    train_pred_df['lgbm_pred'] = oof_preds\n    to_pickle(f'../output/{HOST}_{NB}_lgbm_models.pkl', lgbm_models)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.474627Z","iopub.execute_input":"2023-02-16T15:28:21.475426Z","iopub.status.idle":"2023-02-16T15:28:21.483050Z","shell.execute_reply.started":"2023-02-16T15:28:21.475351Z","shell.execute_reply":"2023-02-16T15:28:21.482294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LightGBM推論","metadata":{}},{"cell_type":"code","source":"# モデル読み込み\nif is_kaggle:\n    lgbm_models = unpickle(f'/kaggle/input/nfl-models/{HOST}_{NB}_lgbm_models.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:21.484168Z","iopub.execute_input":"2023-02-16T15:28:21.485351Z","iopub.status.idle":"2023-02-16T15:28:22.775629Z","shell.execute_reply.started":"2023-02-16T15:28:21.485273Z","shell.execute_reply":"2023-02-16T15:28:22.774595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 推論\ny_lgbm_pred = np.zeros(len(test_pred_df))\nfor model in lgbm_models:\n    #y_lgbm_pred += model.predict(Pool(test_pred_df[features]), prediction_type='Probability')[:, 1] / len(lgbm_models)\n    y_lgbm_pred += model.predict(test_pred_df[features]) / len(lgbm_models)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:22.777152Z","iopub.execute_input":"2023-02-16T15:28:22.777565Z","iopub.status.idle":"2023-02-16T15:28:33.283103Z","shell.execute_reply.started":"2023-02-16T15:28:22.777507Z","shell.execute_reply":"2023-02-16T15:28:33.282024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_df['lgbm_pred'] = y_lgbm_pred","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:33.284815Z","iopub.execute_input":"2023-02-16T15:28:33.285226Z","iopub.status.idle":"2023-02-16T15:28:33.291790Z","shell.execute_reply.started":"2023-02-16T15:28:33.285187Z","shell.execute_reply":"2023-02-16T15:28:33.290754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_kaggle:\n    to_pickle(f'../output/{HOST}_{NB}_test_pred_lgbm_df.pkl', test_pred_df)\nelse:\n    _test_df = unpickle(f'/kaggle/input/nfl-models/{HOST}_{NB}_test_pred_lgbm_df.pkl')\n    \n    if len(test_pred_df) == len(_test_df):#subのときは動かないように\n        plt.figure(figsize=(10, 10))\n        plt.scatter(test_pred_df['lgbm_pred'], _test_df['lgbm_pred'], s=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:33.293267Z","iopub.execute_input":"2023-02-16T15:28:33.295150Z","iopub.status.idle":"2023-02-16T15:28:33.861663Z","shell.execute_reply.started":"2023-02-16T15:28:33.295113Z","shell.execute_reply":"2023-02-16T15:28:33.860575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 閾値値探索","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:28:33.863287Z","iopub.execute_input":"2023-02-16T15:28:33.863687Z","iopub.status.idle":"2023-02-16T15:28:34.038487Z","shell.execute_reply.started":"2023-02-16T15:28:33.863648Z","shell.execute_reply":"2023-02-16T15:28:34.037286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nprint(0)\n\n\ndef opt_thr_2(window_pair, window_g, n_trials=100, log_level=optuna.logging.WARNING):\n    train_pred_ma_pair_df['pred_ma'] = train_pred_ma_pair_df.groupby(['game_play', 'nfl_player_pair'])['lgbm_pred'].rolling(window_pair, center=True, min_periods=1).mean().to_frame('pred_ma').reset_index()['pred_ma']\n    pair_np = train_pred_ma_pair_df['pred_ma'].to_numpy()\n    \n    train_pred_ma_g_df['pred_ma'] = train_pred_ma_g_df.groupby(['game_play', 'nfl_player_pair'])['lgbm_pred'].rolling(window_g, center=True, min_periods=1).mean().to_frame('pred_ma').reset_index()['pred_ma']\n    g_np= train_pred_ma_g_df['pred_ma'].to_numpy()\n    \n    t = np.hstack([train_pred_ma_pair_df['contact'].to_numpy(), train_pred_ma_g_df['contact'].to_numpy()])\n    \n    def calc_score_pair(thr_pair, thr_g):\n        \n        pair_contact_pred = np.where(pair_np > thr_pair, 1, 0)\n        g_contact_pred = np.where(g_np > thr_g, 1, 0)\n\n        score = matthews_corrcoef(t, np.hstack([pair_contact_pred, g_contact_pred]))\n\n        return score\n\n    def objective_pair(trial):\n\n        thr_pair = trial.suggest_uniform('thr_pair', 1e-10, 5e-1)\n        thr_g = trial.suggest_uniform('thr_g', 1e-10, 5e-1)\n\n        score = calc_score_pair(thr_pair, thr_g)\n        return score\n\n    optuna.logging.set_verbosity(log_level)\n\n    pair_study = optuna.create_study(direction='maximize')\n\n    pair_study.optimize(objective_pair, n_trials=n_trials, n_jobs=10)\n    \n    return pair_study\n\nprint(1)\n\nif not is_kaggle:\n    \n    print(2)\n    \n    train_pred_ma_pair_df = train_pred_df[~(train_pred_df['nfl_player_pair'].apply(lambda x : x.endswith('G')))].reset_index(drop=True)\n    train_pred_ma_g_df = train_pred_df[train_pred_df['nfl_player_pair'].apply(lambda x : x.endswith('G'))].reset_index(drop=True)\n    \n    print(3)\n    \n    opt_result = []\n\n    for window_pair in range(1, 7):\n        for window_g in range(1, 20):\n            print(f'{window_pair}, {window_g}')\n            study = opt_thr_2(window_pair, window_g,n_trials=100)\n            print(study.best_params['thr_pair'], study.best_params['thr_g'], study.best_value)\n            opt_result.append({\n                'window_pair':window_pair,\n                'window_g':window_g,\n                'thr_pair':study.best_params['thr_pair'],\n                'thr_g':study.best_params['thr_g'],\n                'score':study.best_value\n            })\n\n    opt_result_df = pd.DataFrame(opt_result)\n    display(opt_result_df.sort_values('score'))\n    print(opt_result_df.sort_values('score').iloc[-1]['thr_pair'])\n    print(opt_result_df.sort_values('score').iloc[-1]['thr_g'])\n    \n    window_pair_value = opt_result_df.sort_values('score').iloc[-1]['window_pair']\n    window_g_value = opt_result_df.sort_values('score').iloc[-1]['window_g']\n    thr_pair_value = opt_result_df.sort_values('score').iloc[-1]['thr_pair']\n    thr_g_value = opt_result_df.sort_values('score').iloc[-1]['thr_g']\n    \n    opt_params = {'window_pair':window_pair_value, 'window_g':window_g_value, 'thr_pair':thr_pair_value, 'thr_g':thr_g_value}\n    \n    to_pickle(f'../output/{HOST}_{NB}_opt_param.pkl', opt_params)","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:35.417065Z","iopub.execute_input":"2023-02-16T15:33:35.417558Z","iopub.status.idle":"2023-02-16T15:33:35.432934Z","shell.execute_reply.started":"2023-02-16T15:33:35.417500Z","shell.execute_reply":"2023-02-16T15:33:35.431722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"exp0126(3080)のoptunaなしの探索\n1, 1\n0.3809473191154542 0.4089767776348963 0.7855585450258817\n1, 2\n0.40055480395573256 0.3325858955155669 0.7859135338863715\n1, 3\n0.40790025686671944 0.38511753637882173 0.7860196737292667\n1, 4\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:36.990495Z","iopub.execute_input":"2023-02-16T15:33:36.993128Z","iopub.status.idle":"2023-02-16T15:33:36.999722Z","shell.execute_reply.started":"2023-02-16T15:33:36.993089Z","shell.execute_reply":"2023-02-16T15:33:36.998647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_kaggle:\n    opt_params = unpickle(f'/kaggle/input/nfl-models/{HOST}_{NB}_opt_param.pkl')\n    \nopt_params","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:37.250104Z","iopub.execute_input":"2023-02-16T15:33:37.251273Z","iopub.status.idle":"2023-02-16T15:33:37.266642Z","shell.execute_reply.started":"2023-02-16T15:33:37.251222Z","shell.execute_reply":"2023-02-16T15:33:37.265749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_ma_pair_df = test_pred_df[~(test_pred_df['nfl_player_pair'].apply(lambda x : x.endswith('G')))].reset_index(drop=True)\ntest_pred_ma_g_df = test_pred_df[test_pred_df['nfl_player_pair'].apply(lambda x : x.endswith('G'))].reset_index(drop=True)\ntest_pred_ma_pair_df['pred_ma'] = test_pred_ma_pair_df.groupby(['game_play', 'nfl_player_pair'])['lgbm_pred'].rolling(int(opt_params['window_pair']), center=True, min_periods=1).mean().to_frame('pred_ma').reset_index()['pred_ma']\ntest_pred_ma_g_df['pred_ma'] = test_pred_ma_g_df.groupby(['game_play', 'nfl_player_pair'])['lgbm_pred'].rolling(int(opt_params['window_g']), center=True, min_periods=1).mean().to_frame('pred_ma').reset_index()['pred_ma']","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:37.653953Z","iopub.execute_input":"2023-02-16T15:33:37.655855Z","iopub.status.idle":"2023-02-16T15:33:37.774306Z","shell.execute_reply.started":"2023-02-16T15:33:37.655810Z","shell.execute_reply":"2023-02-16T15:33:37.773196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_ma_pair_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:38.302991Z","iopub.execute_input":"2023-02-16T15:33:38.303802Z","iopub.status.idle":"2023-02-16T15:33:38.338153Z","shell.execute_reply.started":"2023-02-16T15:33:38.303746Z","shell.execute_reply":"2023-02-16T15:33:38.337013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_ma_g_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:38.981654Z","iopub.execute_input":"2023-02-16T15:33:38.982039Z","iopub.status.idle":"2023-02-16T15:33:39.010594Z","shell.execute_reply.started":"2023-02-16T15:33:38.982009Z","shell.execute_reply":"2023-02-16T15:33:39.009676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thr_pair = opt_params['thr_pair']\nthr_g = opt_params['thr_g']\ntest_pred_ma_pair_df['contact'] = 0\ntest_pred_ma_pair_df.loc[test_pred_ma_pair_df['pred_ma'] > thr_pair, 'contact'] = 1\n\ntest_pred_ma_g_df['contact'] = 0\ntest_pred_ma_g_df.loc[test_pred_ma_g_df['pred_ma'] > thr_g, 'contact'] = 1\n\nsub_df = pd.concat([test_pred_ma_pair_df, test_pred_ma_g_df]).reset_index(drop=True)\nsub_df[['contact_id', 'contact']].to_csv('submission.csv', index=False)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:39.322058Z","iopub.execute_input":"2023-02-16T15:33:39.322411Z","iopub.status.idle":"2023-02-16T15:33:39.467572Z","shell.execute_reply.started":"2023-02-16T15:33:39.322380Z","shell.execute_reply":"2023-02-16T15:33:39.466470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ffmpeg -version","metadata":{"execution":{"iopub.status.busy":"2023-02-16T15:33:40.220819Z","iopub.execute_input":"2023-02-16T15:33:40.221567Z","iopub.status.idle":"2023-02-16T15:33:41.266194Z","shell.execute_reply.started":"2023-02-16T15:33:40.221510Z","shell.execute_reply":"2023-02-16T15:33:41.264765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}