{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nsys.path.append('/kaggle/input/timm-0-6-9/pytorch-image-models-master')\nimport glob\nimport numpy as np\nimport pandas as pd\nimport random\nimport math\nimport gc\nimport cv2\nfrom tqdm import tqdm\nimport time\nfrom functools import lru_cache\nimport torch\nfrom torch import nn\nfrom torch.nn import functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda.amp import autocast, GradScaler\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import matthews_corrcoef","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-26T20:52:32.227486Z","iopub.execute_input":"2023-02-26T20:52:32.227980Z","iopub.status.idle":"2023-02-26T20:52:39.715758Z","shell.execute_reply.started":"2023-02-26T20:52:32.227884Z","shell.execute_reply":"2023-02-26T20:52:39.714269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG = {\n    'seed': 42,\n    'model': 'resnet50',\n    'img_size': 256,\n    'epochs': 10,\n    'train_bs': 100, \n    'valid_bs': 64,\n    'lr': 1e-3, \n    'weight_decay': 1e-6,\n    'num_workers': 2\n}","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:52:39.718943Z","iopub.execute_input":"2023-02-26T20:52:39.720307Z","iopub.status.idle":"2023-02-26T20:52:39.728233Z","shell.execute_reply.started":"2023-02-26T20:52:39.720250Z","shell.execute_reply":"2023-02-26T20:52:39.726708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nseed_everything(CFG['seed'])\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:52:39.730100Z","iopub.execute_input":"2023-02-26T20:52:39.735028Z","iopub.status.idle":"2023-02-26T20:52:39.816000Z","shell.execute_reply.started":"2023-02-26T20:52:39.734971Z","shell.execute_reply":"2023-02-26T20:52:39.814659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def expand_contact_id(df):\n    \"\"\"\n    Splits out contact_id into seperate columns.\n    \"\"\"\n    df[\"game_play\"] = df[\"contact_id\"].str[:12]\n    df[\"step\"] = df[\"contact_id\"].str.split(\"_\").str[-3].astype(\"int\")\n    df[\"nfl_player_id_1\"] = df[\"contact_id\"].str.split(\"_\").str[-2]\n    df[\"nfl_player_id_2\"] = df[\"contact_id\"].str.split(\"_\").str[-1]\n    return df\n\nlabels = expand_contact_id(pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/sample_submission.csv\"))\n\ntest_tracking = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv\")\n\ntest_helmets = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv\")\n\ntest_video_metadata = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_video_metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:52:39.819899Z","iopub.execute_input":"2023-02-26T20:52:39.820441Z","iopub.status.idle":"2023-02-26T20:52:40.712636Z","shell.execute_reply.started":"2023-02-26T20:52:39.820390Z","shell.execute_reply":"2023-02-26T20:52:40.711298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p ../work/frames\n\nfor video in tqdm(test_helmets.video.unique()):\n    if 'Endzone2' not in video:\n        !ffmpeg -i /kaggle/input/nfl-player-contact-detection/test/{video} -q:v 2 -f image2 /kaggle/work/frames/{video}_%04d.jpg -hide_banner -loglevel error","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:52:40.715359Z","iopub.execute_input":"2023-02-26T20:52:40.717015Z","iopub.status.idle":"2023-02-26T20:53:50.534825Z","shell.execute_reply.started":"2023-02-26T20:52:40.716954Z","shell.execute_reply":"2023-02-26T20:53:50.533072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_features(df, tr_tracking, merge_col=\"step\", use_cols=[\"x_position\", \"y_position\"]):\n    output_cols = []\n    df_combo = (\n        df.astype({\"nfl_player_id_1\": \"str\"})\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\",] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_1\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .rename(columns={c: c+\"_1\" for c in use_cols})\n        .drop(\"nfl_player_id\", axis=1)\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\"] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_2\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .drop(\"nfl_player_id\", axis=1)\n        .rename(columns={c: c+\"_2\" for c in use_cols})\n        .sort_values([\"game_play\", merge_col, \"nfl_player_id_1\", \"nfl_player_id_2\"])\n        .reset_index(drop=True)\n    )\n    output_cols += [c+\"_1\" for c in use_cols]\n    output_cols += [c+\"_2\" for c in use_cols]\n    \n    if (\"x_position\" in use_cols) & (\"y_position\" in use_cols):\n        index = df_combo['x_position_2'].notnull()\n        \n        distance_arr = np.full(len(index), np.nan)\n        tmp_distance_arr = np.sqrt(\n            np.square(df_combo.loc[index, \"x_position_1\"] - df_combo.loc[index, \"x_position_2\"])\n            + np.square(df_combo.loc[index, \"y_position_1\"]- df_combo.loc[index, \"y_position_2\"])\n        )\n        \n        distance_arr[index] = tmp_distance_arr\n        df_combo['distance'] = distance_arr\n        output_cols += [\"distance\"]\n        \n    df_combo['G_flug'] = (df_combo['nfl_player_id_2']==\"G\")\n    output_cols += [\"G_flug\"]\n    return df_combo, output_cols\n\n\nuse_cols = [\n    'x_position', 'y_position', 'speed', 'distance',\n    'direction', 'orientation', 'acceleration', 'sa'\n]\n\ntest, feature_cols = create_features(labels, test_tracking, use_cols=use_cols)\ntest","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:50.537403Z","iopub.execute_input":"2023-02-26T20:53:50.538151Z","iopub.status.idle":"2023-02-26T20:53:51.005827Z","shell.execute_reply.started":"2023-02-26T20:53:50.538074Z","shell.execute_reply":"2023-02-26T20:53:51.004346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_filtered = test.query('not distance>2').reset_index(drop=True)\ntest_filtered['frame'] = (test_filtered['step']/10*59.94+5*59.94).astype('int')+1\ntest_filtered","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.010738Z","iopub.execute_input":"2023-02-26T20:53:51.011344Z","iopub.status.idle":"2023-02-26T20:53:51.092566Z","shell.execute_reply.started":"2023-02-26T20:53:51.011191Z","shell.execute_reply":"2023-02-26T20:53:51.091097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test, labels, test_tracking\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.094563Z","iopub.execute_input":"2023-02-26T20:53:51.096603Z","iopub.status.idle":"2023-02-26T20:53:51.335450Z","shell.execute_reply.started":"2023-02-26T20:53:51.096540Z","shell.execute_reply":"2023-02-26T20:53:51.332109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_aug = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.ShiftScaleRotate(p=0.5),\n    A.RandomBrightnessContrast(brightness_limit=(-0.1, 0.1), contrast_limit=(-0.1, 0.1), p=0.5),\n    A.Normalize(mean=[0.], std=[1.]),\n    ToTensorV2()\n])\n\nvalid_aug = A.Compose([\n    A.Normalize(mean=[0.], std=[1.]),\n    ToTensorV2()\n])","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.342716Z","iopub.execute_input":"2023-02-26T20:53:51.349641Z","iopub.status.idle":"2023-02-26T20:53:51.365034Z","shell.execute_reply.started":"2023-02-26T20:53:51.349585Z","shell.execute_reply":"2023-02-26T20:53:51.362912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video2helmets = {}\ntest_helmets_new = test_helmets.set_index('video')\nfor video in tqdm(test_helmets.video.unique()):\n    video2helmets[video] = test_helmets_new.loc[video].reset_index(drop=True)\n    \ndel test_helmets, test_helmets_new\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.373464Z","iopub.execute_input":"2023-02-26T20:53:51.374676Z","iopub.status.idle":"2023-02-26T20:53:51.674811Z","shell.execute_reply.started":"2023-02-26T20:53:51.374615Z","shell.execute_reply":"2023-02-26T20:53:51.673357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video2frames = {}\n\nfor game_play in tqdm(test_video_metadata.game_play.unique()):\n    for view in ['Endzone', 'Sideline']:\n        video = game_play + f'_{view}.mp4'\n        video2frames[video] = max(list(map(lambda x:int(x.split('_')[-1].split('.')[0]), \\\n                                           glob.glob(f'/kaggle/work/frames/{video}*'))))","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.677139Z","iopub.execute_input":"2023-02-26T20:53:51.678153Z","iopub.status.idle":"2023-02-26T20:53:51.771382Z","shell.execute_reply.started":"2023-02-26T20:53:51.678089Z","shell.execute_reply":"2023-02-26T20:53:51.769920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyDataset(Dataset):\n    def __init__(self, df, aug=valid_aug, mode='train'):\n        self.df = df\n        self.frame = df.frame.values\n        self.feature = df[feature_cols].fillna(-1).values\n        self.players = df[['nfl_player_id_1','nfl_player_id_2']].values\n        self.game_play = df.game_play.values\n        self.aug = aug\n        self.mode = mode\n        \n    def __len__(self):\n        return len(self.df)\n    \n    # @lru_cache(1024)\n    # def read_img(self, path):\n    #     return cv2.imread(path, 0)\n   \n    def __getitem__(self, idx):   \n        window = 24\n        frame = self.frame[idx]\n        \n        if self.mode == 'train':\n            frame = frame + random.randint(-6, 6)\n\n        players = []\n        for p in self.players[idx]:\n            if p == 'G':\n                players.append(p)\n            else:\n                players.append(int(p))\n        \n        imgs = []\n        for view in ['Endzone', 'Sideline']:\n            video = self.game_play[idx] + f'_{view}.mp4'\n\n            tmp = video2helmets[video]\n#             tmp = tmp.query('@frame-@window<=frame<=@frame+@window')\n            tmp[tmp['frame'].between(frame-window, frame+window)]\n            tmp = tmp[tmp.nfl_player_id.isin(players)]#.sort_values(['nfl_player_id', 'frame'])\n            tmp_frames = tmp.frame.values\n            tmp = tmp.groupby('frame')[['left','width','top','height']].mean()\n#0.002s\n\n            bboxes = []\n            for f in range(frame-window, frame+window+1, 1):\n                if f in tmp_frames:\n                    x, w, y, h = tmp.loc[f][['left','width','top','height']]\n                    bboxes.append([x, w, y, h])\n                else:\n                    bboxes.append([np.nan, np.nan, np.nan, np.nan])\n            bboxes = pd.DataFrame(bboxes).interpolate(limit_direction='both').values\n            bboxes = bboxes[::4]\n\n            if bboxes.sum() > 0:\n                flag = 1\n            else:\n                flag = 0\n#0.03s\n                    \n            for i, f in enumerate(range(frame-window, frame+window+1, 4)):\n                img_new = np.zeros((256, 256), dtype=np.float32)\n\n                if flag == 1 and f <= video2frames[video]:\n                    img = cv2.imread(f'/kaggle/work/frames/{video}_{f:04d}.jpg', 0)\n\n                    x, w, y, h = bboxes[i]\n\n                    img = img[int(y+h/2)-128:int(y+h/2)+128,int(x+w/2)-128:int(x+w/2)+128].copy()\n                    img_new[:img.shape[0], :img.shape[1]] = img\n                    \n                imgs.append(img_new)\n#0.06s\n                \n        feature = np.float32(self.feature[idx])\n\n        img = np.array(imgs).transpose(1, 2, 0)    \n        img = self.aug(image=img)[\"image\"]\n        label = np.float32(self.df.contact.values[idx])\n\n        return img, feature, label","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.778452Z","iopub.execute_input":"2023-02-26T20:53:51.779595Z","iopub.status.idle":"2023-02-26T20:53:51.816034Z","shell.execute_reply.started":"2023-02-26T20:53:51.779535Z","shell.execute_reply":"2023-02-26T20:53:51.814395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, feature, label = MyDataset(test_filtered, valid_aug, 'test')[0]\nplt.imshow(img.permute(1,2,0)[:,:,7])\nplt.show()\nimg.shape, feature, label","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:51.825540Z","iopub.execute_input":"2023-02-26T20:53:51.829984Z","iopub.status.idle":"2023-02-26T20:53:52.428543Z","shell.execute_reply.started":"2023-02-26T20:53:51.829916Z","shell.execute_reply":"2023-02-26T20:53:52.427123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super(Model, self).__init__()\n        self.backbone = timm.create_model(CFG['model'], pretrained=False, num_classes=500, in_chans=13)\n        self.mlp = nn.Sequential(\n            nn.Linear(18, 64),\n            nn.LayerNorm(64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            # nn.Linear(64, 64),\n            # nn.LayerNorm(64),\n            # nn.ReLU(),\n            # nn.Dropout(0.2)\n        )\n        self.fc = nn.Linear(64+500*2, 1)\n\n    def forward(self, img, feature):\n        b, c, h, w = img.shape\n        img = img.reshape(b*2, c//2, h, w)\n        img = self.backbone(img).reshape(b, -1)\n        feature = self.mlp(feature)\n        y = self.fc(torch.cat([img, feature], dim=1))\n        return y","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:52.430232Z","iopub.execute_input":"2023-02-26T20:53:52.430934Z","iopub.status.idle":"2023-02-26T20:53:52.446382Z","shell.execute_reply.started":"2023-02-26T20:53:52.430879Z","shell.execute_reply":"2023-02-26T20:53:52.444671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = MyDataset(test_filtered, valid_aug, 'test')\ntest_loader = DataLoader(test_set, batch_size=CFG['valid_bs'], shuffle=False, num_workers=CFG['num_workers'], pin_memory=True)\n\nmodel = Model().to(device)\nmodel.load_state_dict(torch.load('/kaggle/input/nfl-exp1/resnet50_fold0.pt'))\n\nmodel.eval()\n    \ny_pred = []\nwith torch.no_grad():\n    tk = tqdm(test_loader, total=len(test_loader))\n    for step, batch in enumerate(tk):\n        if(step % 4 != 3):\n            img, feature, label = [x.to(device) for x in batch]\n            output1 = model(img, feature).squeeze(-1)\n            output2 = model(img.flip(-1), feature).squeeze(-1)\n            \n            y_pred.extend(0.15*(output1.sigmoid().cpu().numpy()) + 0.85*(output2.sigmoid().cpu().numpy()))\n        else:\n            img, feature, label = [x.to(device) for x in batch]\n            output = model(img.flip(-1), feature).squeeze(-1)\n            y_pred.extend(output.sigmoid().cpu().numpy())    \n\ny_pred = np.array(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T20:53:52.449132Z","iopub.execute_input":"2023-02-26T20:53:52.450462Z","iopub.status.idle":"2023-02-26T21:19:55.202989Z","shell.execute_reply.started":"2023-02-26T20:53:52.450401Z","shell.execute_reply":"2023-02-26T21:19:55.200960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_filtered['contact'] = y_pred\n\nsub = pd.read_csv('/kaggle/input/nfl-player-contact-detection/sample_submission.csv')\n\nsub = sub.drop(\"contact\", axis=1).merge(test_filtered[['contact_id', 'contact']], how='left', on='contact_id')\nsub['contact'] = sub['contact'].fillna(0)#.astype('int')\n\nsub[[\"contact_id\", \"contact\"]].to_csv(\"cnn_output.csv\", index=False)\n\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:19:55.205766Z","iopub.execute_input":"2023-02-26T21:19:55.206692Z","iopub.status.idle":"2023-02-26T21:19:55.413416Z","shell.execute_reply.started":"2023-02-26T21:19:55.206634Z","shell.execute_reply":"2023-02-26T21:19:55.412026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a=list(globals().keys())\nfor key in a:\n    if not key.startswith(\"__\"):\n        globals().pop(key)\n\nimport gc,psutil,os\nfrom numba import cuda\ncuda.select_device(0)\ncuda.close()\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:19:55.415478Z","iopub.execute_input":"2023-02-26T21:19:55.416071Z","iopub.status.idle":"2023-02-26T21:19:57.211209Z","shell.execute_reply.started":"2023-02-26T21:19:55.416012Z","shell.execute_reply":"2023-02-26T21:19:57.209569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import psutil\nimport os\nimport gc\nimport subprocess\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom IPython.display import Video, display\n\nfrom scipy.optimize import minimize\nimport cv2\nfrom glob import glob\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import (\n    roc_auc_score,\n    matthews_corrcoef,\n)\n\nimport xgboost as xgb\n\nimport torch\n\nif torch.cuda.is_available():\n    import cupy \n    import cudf\n    from cuml import ForestInference\n\ndef setup(cfg):\n    cfg.device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    \n    # set dirs\n    cfg.INPUT = f'../input/{cfg.COMPETITION}'\n    cfg.EXP = cfg.NAME\n    cfg.OUTPUT_EXP = cfg.NAME\n    cfg.SUBMISSION = './'\n    cfg.DATASET = '../input/'\n\n    cfg.EXP_MODEL = os.path.join(cfg.EXP, 'model')\n    cfg.EXP_FIG = os.path.join(cfg.EXP, 'fig')\n    cfg.EXP_PREDS = os.path.join(cfg.EXP, 'preds')\n\n    # make dirs\n    for d in [cfg.EXP_MODEL, cfg.EXP_FIG, cfg.EXP_PREDS]:\n        os.makedirs(d, exist_ok=True)\n        \n    return cfg\n\nclass Config:\n    AUTHOR = \"colum2131\"\n\n    NAME = \"NFLC-\" + \"Exp001-simple-xgb-baseline\"\n\n    COMPETITION = \"nfl-player-contact-detection\"\n\n    seed = 42\n    num_fold = 16\n   \n    xgb_params = {\n        'objective': 'binary:logistic',\n        'eval_metric': 'auc',\n        'learning_rate':0.005,\n        'tree_method':'hist' if not torch.cuda.is_available() else 'gpu_hist',\n        \"max_depth\":10,\n        \"subsample\":0.5,\n        \"colsample_bytree\":0.8,\n        \"lambda\":10,\n        \"max_bin\":1024,\n        \"max_delta_step\":1\n    }\ncfg = setup(Config)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:19:57.213827Z","iopub.execute_input":"2023-02-26T21:19:57.214916Z","iopub.status.idle":"2023-02-26T21:19:58.367666Z","shell.execute_reply.started":"2023-02-26T21:19:57.214859Z","shell.execute_reply":"2023-02-26T21:19:58.366314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# function\n# ==============================\n# ref: https://www.kaggle.com/code/robikscube/nfl-player-contact-detection-getting-started\ndef add_contact_id(df):\n    # Create contact ids\n    df[\"contact_id\"] = (\n        df[\"game_play\"]\n        + \"_\"\n        + df[\"step\"].astype(\"str\")\n        + \"_\"\n        + df[\"nfl_player_id_1\"].astype(\"str\")\n        + \"_\"\n        + df[\"nfl_player_id_2\"].astype(\"str\")\n    )\n    return df\n\ndef expand_contact_id(df):\n    \"\"\"\n    Splits out contact_id into seperate columns.\n    \"\"\"\n    df[\"game_play\"] = df[\"contact_id\"].str[:12]\n    df[\"step\"] = df[\"contact_id\"].str.split(\"_\").str[-3].astype(\"int\")\n    df[\"nfl_player_id_1\"] = df[\"contact_id\"].str.split(\"_\").str[-2]\n    df[\"nfl_player_id_2\"] = df[\"contact_id\"].str.split(\"_\").str[-1]\n    return df\n\n# cross validation\ndef get_groupkfold(train, target_col, group_col, n_splits):\n    kf = GroupKFold(n_splits=n_splits)\n    generator = kf.split(train, train[target_col], train[group_col])\n    fold_series = []\n    for fold, (idx_train, idx_valid) in enumerate(generator):\n        fold_series.append(pd.Series(fold, index=idx_valid))\n    fold_series = pd.concat(fold_series).sort_index()\n    return fold_series\n\n# xgboost code\ndef fit_xgboost(cfg, X, y, params, add_suffix=''):\n    \"\"\"\n    xgb_params = {\n        'objective': 'binary:logistic',\n        'eval_metric': 'auc',\n        'learning_rate':0.01,\n        'tree_method':'gpu_hist'\n    }\n    \"\"\"\n    oof_pred = np.zeros(len(y), dtype=np.float32)\n    for fold in sorted(cfg.folds.unique()):\n        if fold == -1: continue\n        idx_train = (cfg.folds!=fold)\n        idx_valid = (cfg.folds==fold)\n        x_train, y_train = X[idx_train], y[idx_train]\n        x_valid, y_valid = X[idx_valid], y[idx_valid]\n        #input(len(x_train.groupby(\"game_play\")))\n        \n        display(pd.Series(y_valid).value_counts())\n\n        xgb_train = xgb.DMatrix(x_train, label=y_train)\n        xgb_valid = xgb.DMatrix(x_valid, label=y_valid)\n        evals = [(xgb_train,'train'),(xgb_valid,'eval')]\n\n        model = xgb.train(\n            params,\n            xgb_train,\n            num_boost_round=10_000,\n            early_stopping_rounds=100,\n            evals=evals,\n            verbose_eval=100,\n        )\n\n        model_path = os.path.join(cfg.EXP_MODEL, f'xgb_fold{fold}{add_suffix}.model')\n        model.save_model(model_path)\n        if not torch.cuda.is_available():\n            model = xgb.Booster().load_model(model_path)\n        else:\n            model = ForestInference.load(model_path, output_class=True, model_type='xgboost')\n        pred_i = model.predict_proba(x_valid)[:, 1]\n        oof_pred[x_valid.index] = pred_i\n        score = round(roc_auc_score(y_valid, pred_i), 5)\n        print(f'Performance of the prediction: {score}\\n')\n        del model; gc.collect(2)\n\n    np.save(os.path.join(cfg.EXP_PREDS, f'oof_pred{add_suffix}'), oof_pred)\n    score = round(roc_auc_score(y, oof_pred), 5)\n    print(f'All Performance of the prediction: {score}')\n    return oof_pred\n\ndef pred_xgboost(X, data_dir, add_suffix=''):\n    models = glob(os.path.join(data_dir, f'xgb_fold*{add_suffix}.model'))\n    if not torch.cuda.is_available():\n         models = [xgb.Booster().load_model(model_path) for model in models]\n    else:\n        models = [ForestInference.load(model, output_class=True, model_type='xgboost') for model in models]\n    preds = np.array([model.predict_proba(X)[:, 1] for model in models])\n    preds = np.mean(preds, axis=0)\n    return preds","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:19:58.369429Z","iopub.execute_input":"2023-02-26T21:19:58.369932Z","iopub.status.idle":"2023-02-26T21:19:58.395524Z","shell.execute_reply.started":"2023-02-26T21:19:58.369883Z","shell.execute_reply":"2023-02-26T21:19:58.394083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# read data\n# ==============================\n\nif not torch.cuda.is_available():\n    tr_tracking = pd.read_csv(os.path.join(cfg.INPUT, 'train_player_tracking.csv'), parse_dates=[\"datetime\"])\n    te_tracking = pd.read_csv(os.path.join(cfg.INPUT, 'test_player_tracking.csv'), parse_dates=[\"datetime\"])\n    sub = pd.read_csv(os.path.join(cfg.INPUT, 'sample_submission.csv'))\n\n    train = pd.read_csv(os.path.join(cfg.INPUT, 'train_labels.csv'), parse_dates=[\"datetime\"])\n    test = expand_contact_id(sub)\n    \nelse:\n    tr_tracking = pd.read_csv(os.path.join(cfg.INPUT, 'train_player_tracking.csv'), parse_dates=[\"datetime\"]).sort_values(\"datetime\")\n    te_tracking = pd.read_csv(os.path.join(cfg.INPUT, 'test_player_tracking.csv'), parse_dates=[\"datetime\"]).sort_values(\"datetime\")\n    sub = pd.read_csv(os.path.join(cfg.INPUT, 'sample_submission.csv'))\n\n    train = cudf.read_csv(os.path.join(cfg.INPUT, 'train_labels.csv'), parse_dates=[\"datetime\"])\n    test = cudf.DataFrame(expand_contact_id(sub))\n\ntrain[\"is_test\"]=0\ntest[\"is_test\"]=1\ntest[\"contact\"]=-1\ntest = add_contact_id(test)\n\ncols = ['x_position','y_position','speed','distance','direction','orientation','acceleration','sa']\n#cols = ['speed','distance','direction','orientation','acceleration','sa']\n#cols = ['speed','distance']\nfeature_cols=[]\n\nmain_data=[tr_tracking,te_tracking]\nfor j in [0,1]:\n    data=main_data[j]\n    data[\"game_play_player_id\"]=data[\"game_play\"]+\"_\"+data[\"nfl_player_id\"].astype(\"str\")\n    a=data.groupby(\"game_play_player_id\")\n    \n    for i in [16,32]:\n        for k in cols:\n            k_name=f\"{k}_mean{i}\"\n            data[k_name]=a[k].rolling(i).mean().reset_index(0,drop=True)\n            feature_cols.append(k_name)\n            k_name=f\"{k}_std{i}\"\n            data[k_name]=a[k].rolling(i).std().reset_index(0,drop=True)\n            feature_cols.append(k_name)\n            k_name=f\"{k}_max{i}\"\n            data[k_name]=a[k].rolling(i).max().reset_index(0,drop=True)\n            feature_cols.append(k_name)\n            k_name=f\"{k}_min{i}\"\n            data[k_name]=a[k].rolling(i).min().reset_index(0,drop=True)\n            feature_cols.append(k_name)\n    main_data[j]=cudf.DataFrame.from_pandas(data)\n    \ntr_tracking,te_tracking=main_data\ndel data,a,main_data\n\n\nfeature_cols=list(set(feature_cols))\ntr_tracking[feature_cols]=tr_tracking[feature_cols].astype(\"float32\")\nte_tracking[feature_cols]=te_tracking[feature_cols].astype(\"float32\")\n\n#print(tr_tracking[feature_cols])\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:19:58.397718Z","iopub.execute_input":"2023-02-26T21:19:58.398632Z","iopub.status.idle":"2023-02-26T21:21:27.360369Z","shell.execute_reply.started":"2023-02-26T21:19:58.398575Z","shell.execute_reply":"2023-02-26T21:21:27.358945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# feature engineering\n# ==============================\ndef create_features(df, tr_tracking, merge_col=\"step\", use_cols=[\"x_position\", \"y_position\"]):\n    output_cols = []\n    df_combo = (\n        df.astype({\"nfl_player_id_1\": \"str\"})\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\",] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_1\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .rename(columns={c: c+\"_1\" for c in use_cols})\n        .drop(\"nfl_player_id\", axis=1)\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\"] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_2\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .drop(\"nfl_player_id\", axis=1)\n        .rename(columns={c: c+\"_2\" for c in use_cols})\n        .sort_values([\"game_play\", merge_col, \"nfl_player_id_1\", \"nfl_player_id_2\"])\n        .reset_index(drop=True)\n    )\n    output_cols += [c+\"_1\" for c in use_cols]\n    output_cols += [c+\"_2\" for c in use_cols]\n    \n    if (\"x_position\" in use_cols) & (\"y_position\" in use_cols):\n        index = df_combo['x_position_2'].notnull()\n        if torch.cuda.is_available():\n            index = index.to_array()\n        distance_arr = np.full(len(index), np.nan)\n        tmp_distance_arr = np.sqrt(\n            np.square(df_combo.loc[index, \"x_position_1\"] - df_combo.loc[index, \"x_position_2\"])\n            + np.square(df_combo.loc[index, \"y_position_1\"]- df_combo.loc[index, \"y_position_2\"])\n        )\n        if torch.cuda.is_available():\n            tmp_distance_arr = tmp_distance_arr.to_array()\n        distance_arr[index] = tmp_distance_arr\n        df_combo['distance'] = distance_arr\n        output_cols += [\"distance\"]\n        \n    df_combo['G_flug'] = (df_combo['nfl_player_id_2']==\"G\")\n    output_cols += [\"G_flug\"]\n    return df_combo, output_cols\n\n\nuse_cols = [\n    'x_position', 'y_position', 'speed', 'distance',\n    'direction', 'orientation', 'acceleration', 'sa'\n]+feature_cols\ntrain, feature_cols = create_features(train, tr_tracking, use_cols=use_cols)\ntest, feature_cols = create_features(test, te_tracking, use_cols=use_cols)\n\ndel tr_tracking,te_tracking\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:21:27.362674Z","iopub.execute_input":"2023-02-26T21:21:27.363362Z","iopub.status.idle":"2023-02-26T21:21:30.441576Z","shell.execute_reply.started":"2023-02-26T21:21:27.363312Z","shell.execute_reply":"2023-02-26T21:21:30.440127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exclude distance > 2\nif the distance between two players is greater than 2 then the probability of contact is so low, we will consider it = 0, training data will be reduced from 4.7 M rows to 660 K","metadata":{}},{"cell_type":"code","source":"DISTANCE_THRESH = 2\ntrain=train.sort_values(\"contact_id\")\ntest=test.sort_values(\"contact_id\")\n\ntrain_y = train['contact'].to_pandas().copy().values\noof_pred = np.zeros(len(train))\ncond_dis_train = (train['distance'].to_pandas()<=DISTANCE_THRESH) | (train['distance'].to_pandas().isna())\ncond_dis_test = (test['distance'].to_pandas()<=DISTANCE_THRESH) | (test['distance'].to_pandas().isna())\n\ntrain = train[cond_dis_train]\ntrain.reset_index(inplace = True, drop = True)\n\nprint('number of train data : ',len(train))\n\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:21:30.443532Z","iopub.execute_input":"2023-02-26T21:21:30.444343Z","iopub.status.idle":"2023-02-26T21:21:31.830475Z","shell.execute_reply.started":"2023-02-26T21:21:30.444289Z","shell.execute_reply":"2023-02-26T21:21:31.827938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helmet track Features","metadata":{}},{"cell_type":"code","source":"helmet_train = cudf.read_csv('/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv')\nhelmet_test = cudf.read_csv('/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv')\n#CLUSTERS = [10, 50, 100, 500]\nCLUSTERS = [8,32,128,512]\n#CLUSTERS = [8,16, 32,64, 128, 256,512]\n\ndef add_step_pct(df, cluster):\n    df['step_pct'] = cluster * (df['step']-min(df['step']))/(max(df['step'])-min(df['step']))\n    df['step_pct'] = df['step_pct'].apply(np.ceil).astype(np.int32)\n    return df\n\nmain_data=[train,test]\nhelmet_data=[helmet_train,helmet_test]\n\nfor i in [0,1]:\n    data=main_data[i].to_pandas()\n    helmet=helmet_data[i].to_pandas()\n    helmet.loc[helmet['view']=='Endzone2','view'] = 'Endzone'\n    helmet.rename(columns = {'frame': 'step'}, inplace = True)\n    \n    for cluster in CLUSTERS:\n        data = data.groupby('game_play').apply(lambda x:add_step_pct(x,cluster))\n        #helmet_cache1=helmet.groupby('game_play').apply(lambda x:add_step_pct(x,cluster))\n        for helmet_view in ['Sideline', 'Endzone']:\n            #helmet_cache2=helmet_cache1[helmet_cache1['view']==helmet_view]\n            ########################\n            helmet_cache2=helmet.groupby('game_play').apply(lambda x:add_step_pct(x,cluster))\n            helmet_cache2=helmet_cache2[helmet_cache2['view']==helmet_view]\n            ########################\n            helmet_cache2['helmet_id'] = helmet_cache2['game_play'] + '_' + helmet_cache2['nfl_player_id'].astype(str) + '_' + helmet_cache2['step_pct'].astype(str)\n\n            helmet_cache2 = helmet_cache2[['helmet_id', 'left', 'width', 'top', 'height']].groupby('helmet_id').mean().reset_index()\n            \n            for player_ind in [1, 2]:\n                data['helmet_id'] = data['game_play'] + '_' + data['nfl_player_id_'+str(player_ind)].astype(str) + \\\n                                        '_' + data['step_pct'].astype(str)\n\n\n                data = data.merge(helmet_cache2, how = 'left')\n\n                data.rename(columns = {i:i+'_'+helmet_view+'_'+str(cluster)+'_'+str(player_ind) for i in ['left', 'width', 'top', 'height']}, inplace = True)\n\n                del data['helmet_id']\n\n                feature_cols += [i+'_'+helmet_view+'_'+str(cluster)+'_'+str(player_ind) for i in ['left', 'width', 'top', 'height']]\n            del helmet_cache2\n            _ = gc.collect(2)\n            \n            \n    cache=data.groupby('game_play')[\"step\"]\n    data[\"step_rate\"]=cache.apply(lambda df:(df-min(df))/(max(df)-min(df)))\n    feature_cols += [\"step_rate\"]\n    main_data[i]=cudf.DataFrame.from_pandas(data)\n\ntrain,test=main_data\n\ndel main_data,helmet_data,data,helmet,helmet_train,helmet_test,cache\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:21:31.835512Z","iopub.execute_input":"2023-02-26T21:21:31.835993Z","iopub.status.idle":"2023-02-26T21:24:34.981418Z","shell.execute_reply.started":"2023-02-26T21:21:31.835955Z","shell.execute_reply":"2023-02-26T21:24:34.978272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# add cnn feature","metadata":{}},{"cell_type":"code","source":"train_cnn_feature = cudf.read_csv('/kaggle/input/zzy-cnn-train-output/zzy_cnn_prid_data.csv')\ntrain_cnn_feature.rename(columns={\"contact\":\"cnn_feature\"}, inplace = True)\n\n\ntrain=train.merge(train_cnn_feature, how = 'left',on=\"contact_id\").sort_values(\"contact_id\").reset_index(drop=True)\nprint(np.corrcoef(train[\"cnn_feature\"].to_pandas(),train[\"contact\"].to_pandas())[0,1])\ncache=train[\"cnn_feature\"].to_pandas().to_numpy()\ncache[torch.randperm(len(train))[:int(len(train)//19)]]=0\ntrain[\"cnn_feature\"]=cache\nprint(np.corrcoef(train[\"cnn_feature\"].to_pandas(),train[\"contact\"].to_pandas())[0,1])\n\ndel train_cnn_feature\n\nif os.path.exists('/kaggle/working/cnn_output.csv'):\n    test_cnn_feature = cudf.read_csv('/kaggle/working/cnn_output.csv')\n    test_cnn_feature.rename(columns={\"contact\":\"cnn_feature\"}, inplace = True)\n    test=test.merge(test_cnn_feature, how = 'left',on=\"contact_id\").sort_values(\"contact_id\").reset_index(drop=True)\n    del test_cnn_feature\nelse :\n    print(\"test_cnn_feature have not load\")\n    \n\n\ntrain[\"cnn_feature\"]=train[\"cnn_feature\"]>0.29\ntest[\"cnn_feature\"]=test[\"cnn_feature\"]>0.29\nfeature_cols += [\"cnn_feature\"]\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:06:18.963762Z","iopub.execute_input":"2023-02-26T22:06:18.964255Z","iopub.status.idle":"2023-02-26T22:06:21.306355Z","shell.execute_reply.started":"2023-02-26T22:06:18.964214Z","shell.execute_reply":"2023-02-26T22:06:21.304949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fill missing values for the ground","metadata":{}},{"cell_type":"code","source":"main_data=[train,test]\nfor i in [0,1]:\n    data=main_data[i].to_pandas()\n    for cluster in CLUSTERS:\n        for helmet_view in ['Sideline', 'Endzone']:\n            data.loc[data['G_flug']==True,'left_'+helmet_view+'_'+str(cluster)+'_2'] = data.loc[data['G_flug']==True,'left_'+helmet_view+'_'+str(cluster)+'_1']\n            data.loc[data['G_flug']==True,'top_'+helmet_view+'_'+str(cluster)+'_2'] = data.loc[data['G_flug']==True,'top_'+helmet_view+'_'+str(cluster)+'_1']\n            data.loc[data['G_flug']==True,'width_'+helmet_view+'_'+str(cluster)+'_2'] = 0\n            data.loc[data['G_flug']==True,'height_'+helmet_view+'_'+str(cluster)+'_2'] = 0\n    main_data[i]=cudf.DataFrame.from_pandas(data)\ntrain,test=main_data","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:06:33.069447Z","iopub.execute_input":"2023-02-26T22:06:33.069900Z","iopub.status.idle":"2023-02-26T22:06:40.173396Z","shell.execute_reply.started":"2023-02-26T22:06:33.069862Z","shell.execute_reply":"2023-02-26T22:06:40.172029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Diffrence & Product features","metadata":{}},{"cell_type":"code","source":"main_data=[train,test]\nfor k in [0,1]:\n    data=main_data[k].to_pandas()\n    cols = [i[:-2] for i in train.columns if i[-2:]=='_1' and i!='nfl_player_id_1']\n    data[[i+'_diff' for i in cols]] = np.abs(data[[i+'_1' for i in cols]].values - data[[i+'_2' for i in cols]].values)\n    feature_cols += [i+'_diff' for i in cols]\n    \n    cols = ['x_position', 'y_position', 'speed', 'distance', 'direction', 'orientation', 'acceleration', 'sa']\n    data[[i+'_prod' for i in cols]] = data[[i+'_1' for i in cols]].values * data[[i+'_2' for i in cols]].values\n    feature_cols += [i+'_prod' for i in cols]\n    main_data[k]=cudf.DataFrame.from_pandas(data)\n\ntrain,test=main_data\n\n#train[[i+'_div' for i in cols]] = train[[i+'_1' for i in cols]].values / train[[i+'_2' for i in cols]].values\n#test[[i+'_div' for i in cols]] = test[[i+'_1' for i in cols]].values / test[[i+'_2' for i in cols]].values\n#feature_cols += [i+'_div' for i in cols]\n\n#train[[i+'_div2' for i in cols]] = train[[i+'_2' for i in cols]].values / train[[i+'_1' for i in cols]].values\n#test[[i+'_div2' for i in cols]] = test[[i+'_2' for i in cols]].values / test[[i+'_1' for i in cols]].values\n#feature_cols += [i+'_div2' for i in cols]#0.65765\n\nfeature_cols=list(set(feature_cols))\nprint('number of features : ',len(feature_cols))\nprint('number of train data : ',len(train))\n_ = gc.collect(2)\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:06:44.796662Z","iopub.execute_input":"2023-02-26T22:06:44.797273Z","iopub.status.idle":"2023-02-26T22:06:52.633917Z","shell.execute_reply.started":"2023-02-26T22:06:44.797219Z","shell.execute_reply":"2023-02-26T22:06:52.632193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_data=cudf.concat((train,test))\n\ndef normalization(x):\n    return (x-x.mean())/(x.std()+1e-9)\n\ndef show(cache):\n    index=cache[:,1].astype(np.float32).argsort()\n    cache=cache[index][::-1]\n    print(cache)\n\ndef get_point(train,feature_cols):\n    cache=[]\n    for key in feature_cols:\n        a=train[key].copy()\n        target=train[\"contact\"][~a.isna()]\n        a=a[~a.isna()]\n        a=np.corrcoef(a,target)[0,1]\n        cache.append([key,abs(a)])\n    cache=np.array(cache)\n    return cache\n\ndef fix_distributed(a):\n    a=normalization(a)\n    skew=(a**3).mean()\n    if skew<0:\n        a=-a\n        skew=-skew\n    i=0\n    while skew>0.6:\n        a=np.log(a-a.min()+1)\n        a=normalization(a)\n        skew=(a**3).mean()\n        if i==10:\n            break\n        i+=1\n        \n    kurt=(a**4).mean()\n    i=0\n    while kurt>3:\n        a=np.tanh(a/2)\n        a=normalization(a)\n        kurt=(a**4).mean()\n        if i==10:\n            break\n        i+=1\n\n    return a\n\n\ncache=get_point(train.to_pandas(),feature_cols)\nprint(cache[:,1].astype(np.float32).sum())\n\nfor key in tqdm(feature_cols):\n    a=main_data[key].to_pandas()\n    main_data[key]=fix_distributed(a).astype(\"float16\")\n\ntrain=main_data[main_data[\"is_test\"]==0]\ntest=main_data[main_data[\"is_test\"]==1]\n\ncache=get_point(train.to_pandas(),feature_cols)\n\nindex=cache[:,1].astype(np.float32).argsort()\npoints=cache[index][::-1]\n\n\ndel main_data,a\n_ = gc.collect(2)\nprint(points[:,1].astype(np.float32).sum())\nprint(psutil.Process(os.getpid()).memory_info().rss/1024/1024)\nprint(points[:30])","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:07:00.082424Z","iopub.execute_input":"2023-02-26T22:07:00.082875Z","iopub.status.idle":"2023-02-26T22:08:01.232659Z","shell.execute_reply.started":"2023-02-26T22:07:00.082838Z","shell.execute_reply":"2023-02-26T22:08:01.231250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train[\"cnn_feature\"]=fix_distributed(train[\"cnn_feature\"].to_pandas()).astype(\"float16\")\n#test[\"cnn_feature\"]=fix_distributed(test[\"cnn_feature\"].to_pandas()).astype(\"float16\")\ntest[\"cnn_feature\"].to_pandas().hist(bins=200)\nplt.show()\ntrain[\"cnn_feature\"].to_pandas().hist(bins=200)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:08:06.195553Z","iopub.execute_input":"2023-02-26T22:08:06.195999Z","iopub.status.idle":"2023-02-26T22:08:07.605642Z","shell.execute_reply.started":"2023-02-26T22:08:06.195962Z","shell.execute_reply":"2023-02-26T22:08:07.604250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train & Infer XGBoost model","metadata":{}},{"cell_type":"code","source":"train=train.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:08:29.444322Z","iopub.execute_input":"2023-02-26T22:08:29.444806Z","iopub.status.idle":"2023-02-26T22:08:32.782113Z","shell.execute_reply.started":"2023-02-26T22:08:29.444769Z","shell.execute_reply":"2023-02-26T22:08:32.780580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================\n# training & inference\n# ==============================\n\ncfg.folds = get_groupkfold(train, 'contact', 'game_play', cfg.num_fold)\ncfg.folds.to_csv(os.path.join(cfg.EXP_PREDS, 'folds.csv'), index=False)\n\noof_pred[np.where(cond_dis_train)] = fit_xgboost(cfg, train[feature_cols], train['contact'], \n                                              cfg.xgb_params, add_suffix=\"_xgb_1st\")\nnp.save('oof_pred.npy',oof_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:08:34.914281Z","iopub.execute_input":"2023-02-26T22:08:34.914738Z","iopub.status.idle":"2023-02-26T22:15:33.445624Z","shell.execute_reply.started":"2023-02-26T22:08:34.914702Z","shell.execute_reply":"2023-02-26T22:15:33.444227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# ==============================\n# optimize\n# ==============================\ndef func(x_list):\n    score = matthews_corrcoef(train_y, oof_pred>x_list[0])\n    return -score\n\nx0 = [0.5]\nresult = minimize(func,x0, method=\"nelder-mead\")\ncfg.threshold = result.x[0]\nprint(\"score:\", round(matthews_corrcoef(train_y, oof_pred>cfg.threshold), 5))\nprint(\"threshold\", round(cfg.threshold, 5))","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:15:58.324992Z","iopub.execute_input":"2023-02-26T22:15:58.325415Z","iopub.status.idle":"2023-02-26T22:16:44.794746Z","shell.execute_reply.started":"2023-02-26T22:15:58.325377Z","shell.execute_reply":"2023-02-26T22:16:44.793237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\n_ = gc.collect(2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pred = pred_xgboost(test.loc[cond_dis_test, feature_cols].to_pandas(), cfg.EXP_MODEL, add_suffix=\"_xgb_1st\")\n\ntest['contact'] = 0\ntest.loc[cond_dis_test, 'contact'] = sub_pred\ntest[['contact_id', 'contact']].to_csv('xgb_output.csv', index=False)\n\ntest['contact'] = (test['contact'] > cfg.threshold).astype(int)\ntest[['contact_id', 'contact']].to_csv('submission.csv', index=False)\ndisplay(test[['contact_id', 'contact']].head())","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:17:23.640425Z","iopub.execute_input":"2023-02-26T22:17:23.640803Z","iopub.status.idle":"2023-02-26T22:17:24.266129Z","shell.execute_reply.started":"2023-02-26T22:17:23.640764Z","shell.execute_reply":"2023-02-26T22:17:24.265129Z"},"trusted":true},"execution_count":null,"outputs":[]}]}