{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":40277,"databundleVersionId":4725531,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":4133932,"sourceType":"datasetVersion","datasetId":2442142},{"sourceId":4768810,"sourceType":"datasetVersion","datasetId":2737246},{"sourceId":10888780,"sourceType":"datasetVersion","datasetId":6766316},{"sourceId":10899323,"sourceType":"datasetVersion","datasetId":6773736}],"dockerImageVersionId":30302,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nsys.path.append('/kaggle/input/timm-0-6-9/pytorch-image-models-master')\nimport glob\nimport numpy as np\nimport pandas as pd\nimport random\nimport math\nimport gc\nimport cv2\nfrom tqdm import tqdm\nimport time\nfrom functools import lru_cache\nimport torch\nfrom torch import nn\nfrom torch.nn import functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda.amp import autocast, GradScaler\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import matthews_corrcoef","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-03-02T09:22:36.748215Z","iopub.execute_input":"2025-03-02T09:22:36.748585Z","iopub.status.idle":"2025-03-02T09:22:42.386412Z","shell.execute_reply.started":"2025-03-02T09:22:36.748553Z","shell.execute_reply":"2025-03-02T09:22:42.385572Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CFG = {\n    'seed': 42,\n    'model': 'resnet18',\n    'img_size': 128,\n    'epochs': 2,\n    'train_bs': 32, \n    'valid_bs': 64,\n    'lr': 1e-3, \n    'weight_decay': 1e-6,\n    'num_workers': 1\n}","metadata":{"execution":{"iopub.status.busy":"2025-03-02T09:22:42.387805Z","iopub.execute_input":"2025-03-02T09:22:42.388290Z","iopub.status.idle":"2025-03-02T09:22:42.393067Z","shell.execute_reply.started":"2025-03-02T09:22:42.388260Z","shell.execute_reply":"2025-03-02T09:22:42.392022Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nseed_everything(CFG['seed'])\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2025-03-02T09:22:42.418141Z","iopub.execute_input":"2025-03-02T09:22:42.418410Z","iopub.status.idle":"2025-03-02T09:22:42.469435Z","shell.execute_reply.started":"2025-03-02T09:22:42.418386Z","shell.execute_reply":"2025-03-02T09:22:42.468689Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def expand_contact_id(df):\n#     \"\"\"\n#     Splits out contact_id into seperate columns.\n#     \"\"\"\n#     df[\"game_play\"] = df[\"contact_id\"].str[:12]\n#     df[\"step\"] = df[\"contact_id\"].str.split(\"_\").str[-3].astype(\"int\")\n#     df[\"nfl_player_id_1\"] = df[\"contact_id\"].str.split(\"_\").str[-2]\n#     df[\"nfl_player_id_2\"] = df[\"contact_id\"].str.split(\"_\").str[-1]\n#     return df\n\n# train_labels = expand_contact_id(pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/train_labels.csv\"))\n\n# train_tracking = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv\")\n\n# train_helmets = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv\")\n\n# train_video_metadata = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/train_video_metadata.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T09:22:45.267439Z","iopub.execute_input":"2025-03-02T09:22:45.267773Z","iopub.status.idle":"2025-03-02T09:23:38.435152Z","shell.execute_reply.started":"2025-03-02T09:22:45.267745Z","shell.execute_reply":"2025-03-02T09:23:38.434397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !mkdir -p ../work/frames\n\n# for video in tqdm(train_helmets.video.unique()):\n#     if 'Endzone2' not in video:\n#         !ffmpeg -i /kaggle/input/nfl-player-contact-detection/train/{video} -q:v 2 -f image2 /kaggle/work/frames/{video}_%04d.jpg -hide_banner -loglevel error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T09:23:38.436595Z","iopub.execute_input":"2025-03-02T09:23:38.436867Z","iopub.status.idle":"2025-03-02T10:08:49.336725Z","shell.execute_reply.started":"2025-03-02T09:23:38.436843Z","shell.execute_reply":"2025-03-02T10:08:49.335559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def create_features(df, tr_tracking, merge_col=\"step\", use_cols=[\"x_position\", \"y_position\"]):\n#     output_cols = []\n#     df_combo = (\n#         df.astype({\"nfl_player_id_1\": \"str\"})\n#         .merge(\n#             tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n#                 [\"game_play\", merge_col, \"nfl_player_id\",] + use_cols\n#             ],\n#             left_on=[\"game_play\", merge_col, \"nfl_player_id_1\"],\n#             right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n#             how=\"left\",\n#         )\n#         .rename(columns={c: c+\"_1\" for c in use_cols})\n#         .drop(\"nfl_player_id\", axis=1)\n#         .merge(\n#             tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n#                 [\"game_play\", merge_col, \"nfl_player_id\"] + use_cols\n#             ],\n#             left_on=[\"game_play\", merge_col, \"nfl_player_id_2\"],\n#             right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n#             how=\"left\",\n#         )\n#         .drop(\"nfl_player_id\", axis=1)\n#         .rename(columns={c: c+\"_2\" for c in use_cols})\n#         .sort_values([\"game_play\", merge_col, \"nfl_player_id_1\", \"nfl_player_id_2\"])\n#         .reset_index(drop=True)\n#     )\n#     output_cols += [c+\"_1\" for c in use_cols]\n#     output_cols += [c+\"_2\" for c in use_cols]\n    \n#     if (\"x_position\" in use_cols) & (\"y_position\" in use_cols):\n#         index = df_combo['x_position_2'].notnull()\n        \n#         distance_arr = np.full(len(index), np.nan)\n#         tmp_distance_arr = np.sqrt(\n#             np.square(df_combo.loc[index, \"x_position_1\"] - df_combo.loc[index, \"x_position_2\"])\n#             + np.square(df_combo.loc[index, \"y_position_1\"]- df_combo.loc[index, \"y_position_2\"])\n#         )\n        \n#         distance_arr[index] = tmp_distance_arr\n#         df_combo['distance'] = distance_arr\n#         output_cols += [\"distance\"]\n        \n#     df_combo['G_flug'] = (df_combo['nfl_player_id_2']==\"G\")\n#     output_cols += [\"G_flug\"]\n#     return df_combo, output_cols\n\n\n# use_cols = [\n#     'x_position', 'y_position', 'speed', 'distance',\n#     'direction', 'orientation', 'acceleration', 'sa'\n# ]\n\n# train, feature_cols = create_features(train_labels, train_tracking, use_cols=use_cols)\n# train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:08:49.338654Z","iopub.execute_input":"2025-03-02T10:08:49.340030Z","iopub.status.idle":"2025-03-02T10:09:20.815007Z","shell.execute_reply.started":"2025-03-02T10:08:49.339992Z","shell.execute_reply":"2025-03-02T10:09:20.814005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_filtered = train.query('not distance>2').reset_index(drop=True)\n# train_filtered['frame'] = (train_filtered['step']/10*59.94+5*59.94).astype('int')+1\n# train_filtered","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:09:20.817027Z","iopub.execute_input":"2025-03-02T10:09:20.817337Z","iopub.status.idle":"2025-03-02T10:09:22.478435Z","shell.execute_reply.started":"2025-03-02T10:09:20.817310Z","shell.execute_reply":"2025-03-02T10:09:22.477390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# del train, train_labels, train_tracking\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:09:22.479581Z","iopub.execute_input":"2025-03-02T10:09:22.479869Z","iopub.status.idle":"2025-03-02T10:09:22.887842Z","shell.execute_reply.started":"2025-03-02T10:09:22.479843Z","shell.execute_reply":"2025-03-02T10:09:22.886682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_aug = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.ShiftScaleRotate(p=0.5),\n    A.RandomBrightnessContrast(brightness_limit=(-0.1, 0.1), contrast_limit=(-0.1, 0.1), p=0.5),\n    A.Normalize(mean=[0.], std=[1.]),\n    ToTensorV2()\n])\n\nvalid_aug = A.Compose([\n    A.Normalize(mean=[0.], std=[1.]),\n    ToTensorV2()\n])","metadata":{"execution":{"iopub.status.busy":"2025-03-02T10:09:22.889511Z","iopub.execute_input":"2025-03-02T10:09:22.889994Z","iopub.status.idle":"2025-03-02T10:09:22.897124Z","shell.execute_reply.started":"2025-03-02T10:09:22.889957Z","shell.execute_reply":"2025-03-02T10:09:22.896247Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# video2helmets = {}\n# train_helmets_new = train_helmets.set_index('video')\n# for video in tqdm(train_helmets.video.unique()):\n#     video2helmets[video] = train_helmets_new.loc[video].reset_index(drop=True)\n    \n# del train_helmets, train_helmets_new\n# gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:09:22.898289Z","iopub.execute_input":"2025-03-02T10:09:22.898550Z","iopub.status.idle":"2025-03-02T10:09:58.012253Z","shell.execute_reply.started":"2025-03-02T10:09:22.898527Z","shell.execute_reply":"2025-03-02T10:09:58.011355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# video2frames = {}\n\n# for game_play in tqdm(train_video_metadata.game_play.unique()):\n#     for view in ['Endzone', 'Sideline']:\n#         video = game_play + f'_{view}.mp4'\n#         video2frames[video] = max(list(map(lambda x:int(x.split('_')[-1].split('.')[0]), \\\n#                                            glob.glob(f'/kaggle/work/frames/{video}*'))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:09:58.013354Z","iopub.execute_input":"2025-03-02T10:09:58.013638Z","iopub.status.idle":"2025-03-02T10:14:11.191810Z","shell.execute_reply.started":"2025-03-02T10:09:58.013611Z","shell.execute_reply":"2025-03-02T10:14:11.190807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class MyDataset(Dataset):\n#     def __init__(self, df, aug=train_aug, mode='train'):\n#         df = df[:len(df)//10]\n#         self.df = df\n#         self.frame = df.frame.values\n#         self.feature = df[feature_cols].fillna(-1).values\n#         self.players = df[['nfl_player_id_1','nfl_player_id_2']].values\n#         self.game_play = df.game_play.values\n#         self.aug = aug\n#         self.mode = mode\n        \n#     def __len__(self):\n#         return len(self.df)\n    \n#     # @lru_cache(1024)\n#     # def read_img(self, path):\n#     #     return cv2.imread(path, 0)\n   \n#     def __getitem__(self, idx):   \n#         window = 24\n#         frame = self.frame[idx]\n        \n#         if self.mode == 'train':\n#             frame = frame + random.randint(-6, 6)\n\n#         players = []\n#         for p in self.players[idx]:\n#             if p == 'G':\n#                 players.append(p)\n#             else:\n#                 players.append(int(p))\n        \n#         imgs = []\n#         for view in ['Endzone', 'Sideline']:\n#             video = self.game_play[idx] + f'_{view}.mp4'\n\n#             tmp = video2helmets[video]\n# #             tmp = tmp.query('@frame-@window<=frame<=@frame+@window')\n#             tmp[tmp['frame'].between(frame-window, frame+window)]\n#             tmp = tmp[tmp.nfl_player_id.isin(players)]#.sort_values(['nfl_player_id', 'frame'])\n#             tmp_frames = tmp.frame.values\n#             tmp = tmp.groupby('frame')[['left','width','top','height']].mean()\n# #0.002s\n\n#             bboxes = []\n#             for f in range(frame-window, frame+window+1, 1):\n#                 if f in tmp_frames:\n#                     x, w, y, h = tmp.loc[f][['left','width','top','height']]\n#                     bboxes.append([x, w, y, h])\n#                 else:\n#                     bboxes.append([np.nan, np.nan, np.nan, np.nan])\n#             bboxes = pd.DataFrame(bboxes).interpolate(limit_direction='both').values\n#             bboxes = bboxes[::4]\n\n#             if bboxes.sum() > 0:\n#                 flag = 1\n#             else:\n#                 flag = 0\n# #0.03s\n                    \n#             for i, f in enumerate(range(frame-window, frame+window+1, 4)):\n#                 img_new = np.zeros((128, 128), dtype=np.float32)\n\n#                 if flag == 1 and f <= video2frames[video]:\n#                     img = cv2.imread(f'/kaggle/work/frames/{video}_{f:04d}.jpg', 0)\n\n#                     x, w, y, h = bboxes[i]\n\n#                     img = img[int(y+h/2)-64:int(y+h/2)+64,int(x+w/2)-64:int(x+w/2)+64].copy()\n#                     img_new[:img.shape[0], :img.shape[1]] = img\n                    \n#                 imgs.append(img_new)\n# #0.06s\n                \n#         feature = np.float32(self.feature[idx])\n\n#         img = np.array(imgs).transpose(1, 2, 0)    \n#         img = self.aug(image=img)[\"image\"]\n#         label = np.float32(self.df.contact.values[idx])\n\n#         return img, feature, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:14:11.193067Z","iopub.execute_input":"2025-03-02T10:14:11.193350Z","iopub.status.idle":"2025-03-02T10:14:11.210822Z","shell.execute_reply.started":"2025-03-02T10:14:11.193324Z","shell.execute_reply":"2025-03-02T10:14:11.209809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# img, feature, label = MyDataset(train_filtered, train_aug, 'train')[0]\n# plt.imshow(img.permute(1,2,0)[:,:,7])\n# plt.show()\n# img.shape, feature, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:14:11.213514Z","iopub.execute_input":"2025-03-02T10:14:11.213791Z","iopub.status.idle":"2025-03-02T10:14:11.844746Z","shell.execute_reply.started":"2025-03-02T10:14:11.213768Z","shell.execute_reply":"2025-03-02T10:14:11.843874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PositionAttention(nn.Module):\n\n    def __init__(self, in_channels):\n        super().__init__()\n        self.query_conv = nn.Conv2d(\n            in_channels, in_channels // 8, kernel_size=1)\n        self.key_conv = nn.Conv2d(in_channels, in_channels // 8, kernel_size=1)\n        self.value_conv = nn.Conv2d(in_channels, in_channels, kernel_size=1)\n        self.gamma = nn.Parameter(torch.zeros(1))\n\n        self.softmax = nn.Softmax(dim=-1)\n\n    def forward(self, x):\n\n        N, C, H, W = x.shape\n        query = self.query_conv(x).view(\n            N, -1, H*W).permute(0, 2, 1)  # (N, H*W, C')\n        key = self.key_conv(x).view(N, -1, H*W)  # (N, C', H*W)\n\n        # caluculate correlation\n        energy = torch.bmm(query, key)    # (N, H*W, H*W)\n        # spatial normalize\n        attention = self.softmax(energy)\n\n        value = self.value_conv(x).view(N, -1, H*W)    # (N, C, H*W)\n\n        out = torch.bmm(value, attention.permute(0, 2, 1))\n        out = out.view(N, C, H, W)\n        out = self.gamma*out + x\n        return out\n\n\n# class ChannelAttention(nn.Module):\n#     def __init__(self):\n#         super().__init__()\n#         self.gamma = nn.Parameter(torch.zeros(1))\n#         self.softmax = nn.Softmax(dim=-1)\n\n#     def forward(self, x):\n\n#         N, C, H, W = x.shape\n#         query = x.view(N, C, -1)    # (N, C, H*W)\n#         key = x.view(N, C, -1).permute(0, 2, 1)    # (N, H*W, C)\n\n#         # calculate correlation\n#         energy = torch.bmm(query, key)    # (N, C, C)\n#         energy = torch.max(\n#             energy, -1, keepdim=True)[0].expand_as(energy) - energy\n#         attention = self.softmax(energy)\n\n#         value = x.view(N, C, -1)\n\n#         out = torch.bmm(attention, value)\n#         out = out.view(N, C, H, W)\n#         out = self.gamma*out + x\n#         return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:14:11.846179Z","iopub.execute_input":"2025-03-02T10:14:11.846851Z","iopub.status.idle":"2025-03-02T10:14:11.858920Z","shell.execute_reply.started":"2025-03-02T10:14:11.846813Z","shell.execute_reply":"2025-03-02T10:14:11.858092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super(Model, self).__init__()\n        self.backbone = timm.create_model(CFG['model'], pretrained=False, num_classes=500, in_chans=13)\n\n        self.feature_reduction = nn.Linear(1000, 512)\n\n        self.position_attention = PositionAttention(512)  # Update input size here\n        #self.channel_attention = ChannelAttention()  # Update input size here\n\n        self.mlp = nn.Sequential(\n            nn.Linear(18, 64),\n            nn.LayerNorm(64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            # nn.Linear(64, 64),\n            # nn.LayerNorm(64),\n            # nn.ReLU(),\n            # nn.Dropout(0.2)\n        )\n        self.fc = nn.Linear(64 + 512, 1)\n\n    def forward(self, img, feature):\n        B, C, H, W = img.shape\n        img = img.reshape(B*2, C//2, H, W)\n        img = self.backbone(img).reshape(B, -1, 1, 1)  # Output is [B, 1000, 1, 1]\n        \n        img = self.feature_reduction(img.view(B, -1))  # Reduce from 1000 -> 512\n        img = img.view(B, 512, 1, 1)  # Reshape back to (B, C, H, W)\n    \n        # Apply position and channel attention in parallel\n        pos_att = self.position_attention(img)\n        #chan_att = self.channel_attention(img)\n    \n        # Fuse outputs (e.g., element-wise sum)\n        #img = pos_att + chan_att  # You can also use concatenation or other fusion methods\n        img = img.view(B, -1)  # Flatten\n        feature = self.mlp(feature) \n        y = self.fc(torch.cat([img, feature], dim=1))\n        \n        return y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:14:11.860032Z","iopub.execute_input":"2025-03-02T10:14:11.860279Z","iopub.status.idle":"2025-03-02T10:14:11.870873Z","shell.execute_reply.started":"2025-03-02T10:14:11.860257Z","shell.execute_reply":"2025-03-02T10:14:11.870156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_set = MyDataset(train_filtered, train_aug, 'train')\n# train_loader = DataLoader(train_set, batch_size=CFG['train_bs'], shuffle=True, num_workers=CFG['num_workers'], pin_memory=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:14:11.872060Z","iopub.execute_input":"2025-03-02T10:14:11.872398Z","iopub.status.idle":"2025-03-02T10:14:11.924353Z","shell.execute_reply.started":"2025-03-02T10:14:11.872365Z","shell.execute_reply":"2025-03-02T10:14:11.923312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = Model().to(device)\n# optimizer = torch.optim.AdamW(model.parameters(), lr=CFG['lr'], weight_decay=CFG['weight_decay'])\n# criterion = nn.BCEWithLogitsLoss()\n\n# scaler = torch.cuda.amp.GradScaler()  # Add this before training\n\n# for epoch in range(CFG['epochs']):\n#     model.train()\n#     running_loss = 0.0\n#     for img, feature, label in tqdm(train_loader):\n#         img, feature, label = img.to(device), feature.to(device), label.to(device).float()\n        \n#         optimizer.zero_grad()\n#         with torch.cuda.amp.autocast():  # Enables mixed precision\n#             output = model(img, feature).squeeze(-1)\n#             loss = criterion(output, label)\n\n#         scaler.scale(loss).backward()\n#         scaler.step(optimizer)\n#         scaler.update()\n        \n#         running_loss += loss.item()\n    \n#     print(f\"Epoch [{epoch+1}/{CFG['epochs']}], Loss: {running_loss/len(train_loader):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T10:14:11.925809Z","iopub.execute_input":"2025-03-02T10:14:11.926126Z","iopub.status.idle":"2025-03-02T16:08:08.195816Z","shell.execute_reply.started":"2025-03-02T10:14:11.926099Z","shell.execute_reply":"2025-03-02T16:08:08.194722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# torch.save(model.state_dict(), \"/kaggle/working/weights_position_only.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:09:29.136614Z","iopub.execute_input":"2025-03-02T16:09:29.137173Z","iopub.status.idle":"2025-03-02T16:09:29.251120Z","shell.execute_reply.started":"2025-03-02T16:09:29.137128Z","shell.execute_reply":"2025-03-02T16:09:29.250174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def expand_contact_id(df):\n    \"\"\"\n    Splits out contact_id into seperate columns.\n    \"\"\"\n    df[\"game_play\"] = df[\"contact_id\"].str[:12]\n    df[\"step\"] = df[\"contact_id\"].str.split(\"_\").str[-3].astype(\"int\")\n    df[\"nfl_player_id_1\"] = df[\"contact_id\"].str.split(\"_\").str[-2]\n    df[\"nfl_player_id_2\"] = df[\"contact_id\"].str.split(\"_\").str[-1]\n    return df\n\nlabels = expand_contact_id(pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/sample_submission.csv\"))\n\ntest_tracking = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv\")\n\ntest_helmets = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv\")\n\ntest_video_metadata = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_video_metadata.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:11:40.888131Z","iopub.execute_input":"2025-03-02T16:11:40.889041Z","iopub.status.idle":"2025-03-02T16:11:41.591532Z","shell.execute_reply.started":"2025-03-02T16:11:40.889007Z","shell.execute_reply":"2025-03-02T16:11:41.590550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p ../work/frames\n\nfor video in tqdm(test_helmets.video.unique()):\n    if 'Endzone2' not in video:\n        !ffmpeg -i /kaggle/input/nfl-player-contact-detection/test/{video} -q:v 2 -f image2 /kaggle/work/frames/{video}_%04d.jpg -hide_banner -loglevel error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:11:43.050661Z","iopub.execute_input":"2025-03-02T16:11:43.051049Z","iopub.status.idle":"2025-03-02T16:12:10.694467Z","shell.execute_reply.started":"2025-03-02T16:11:43.051016Z","shell.execute_reply":"2025-03-02T16:12:10.693271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_features(df, tr_tracking, merge_col=\"step\", use_cols=[\"x_position\", \"y_position\"]):\n    output_cols = []\n    df_combo = (\n        df.astype({\"nfl_player_id_1\": \"str\"})\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\",] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_1\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .rename(columns={c: c+\"_1\" for c in use_cols})\n        .drop(\"nfl_player_id\", axis=1)\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\"] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_2\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .drop(\"nfl_player_id\", axis=1)\n        .rename(columns={c: c+\"_2\" for c in use_cols})\n        .sort_values([\"game_play\", merge_col, \"nfl_player_id_1\", \"nfl_player_id_2\"])\n        .reset_index(drop=True)\n    )\n    output_cols += [c+\"_1\" for c in use_cols]\n    output_cols += [c+\"_2\" for c in use_cols]\n    \n    if (\"x_position\" in use_cols) & (\"y_position\" in use_cols):\n        index = df_combo['x_position_2'].notnull()\n        \n        distance_arr = np.full(len(index), np.nan)\n        tmp_distance_arr = np.sqrt(\n            np.square(df_combo.loc[index, \"x_position_1\"] - df_combo.loc[index, \"x_position_2\"])\n            + np.square(df_combo.loc[index, \"y_position_1\"]- df_combo.loc[index, \"y_position_2\"])\n        )\n        \n        distance_arr[index] = tmp_distance_arr\n        df_combo['distance'] = distance_arr\n        output_cols += [\"distance\"]\n        \n    df_combo['G_flug'] = (df_combo['nfl_player_id_2']==\"G\")\n    output_cols += [\"G_flug\"]\n    return df_combo, output_cols\n\n\nuse_cols = [\n    'x_position', 'y_position', 'speed', 'distance',\n    'direction', 'orientation', 'acceleration', 'sa'\n]\n\ntest, feature_cols = create_features(labels, test_tracking, use_cols=use_cols)\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:10.696586Z","iopub.execute_input":"2025-03-02T16:12:10.696878Z","iopub.status.idle":"2025-03-02T16:12:10.983579Z","shell.execute_reply.started":"2025-03-02T16:12:10.696851Z","shell.execute_reply":"2025-03-02T16:12:10.982633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_filtered = test.query('not distance>2').reset_index(drop=True)\ntest_filtered['frame'] = (test_filtered['step']/10*59.94+5*59.94).astype('int')+1\ntest_filtered","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:10.985004Z","iopub.execute_input":"2025-03-02T16:12:10.985295Z","iopub.status.idle":"2025-03-02T16:12:11.028735Z","shell.execute_reply.started":"2025-03-02T16:12:10.985270Z","shell.execute_reply":"2025-03-02T16:12:11.027848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del test, labels, test_tracking\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:11.030681Z","iopub.execute_input":"2025-03-02T16:12:11.030984Z","iopub.status.idle":"2025-03-02T16:12:11.230332Z","shell.execute_reply.started":"2025-03-02T16:12:11.030957Z","shell.execute_reply":"2025-03-02T16:12:11.229275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"video2helmets = {}\ntest_helmets_new = test_helmets.set_index('video')\nfor video in tqdm(test_helmets.video.unique()):\n    video2helmets[video] = test_helmets_new.loc[video].reset_index(drop=True)\n    \ndel test_helmets, test_helmets_new\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:11.231885Z","iopub.execute_input":"2025-03-02T16:12:11.232777Z","iopub.status.idle":"2025-03-02T16:12:11.448690Z","shell.execute_reply.started":"2025-03-02T16:12:11.232731Z","shell.execute_reply":"2025-03-02T16:12:11.447839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"video2frames = {}\n\nfor game_play in tqdm(test_video_metadata.game_play.unique()):\n    for view in ['Endzone', 'Sideline']:\n        video = game_play + f'_{view}.mp4'\n        video2frames[video] = max(list(map(lambda x:int(x.split('_')[-1].split('.')[0]), \\\n                                           glob.glob(f'/kaggle/work/frames/{video}*'))))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:11.449804Z","iopub.execute_input":"2025-03-02T16:12:11.450104Z","iopub.status.idle":"2025-03-02T16:12:13.561249Z","shell.execute_reply.started":"2025-03-02T16:12:11.450070Z","shell.execute_reply":"2025-03-02T16:12:13.560383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MyTestDataset(Dataset):\n    def __init__(self, df, aug=valid_aug, mode='train'):\n        self.df = df\n        self.frame = df.frame.values\n        self.feature = df[feature_cols].fillna(-1).values\n        self.players = df[['nfl_player_id_1','nfl_player_id_2']].values\n        self.game_play = df.game_play.values\n        self.aug = aug\n        self.mode = mode\n        \n    def __len__(self):\n        return len(self.df)\n    \n    # @lru_cache(1024)\n    # def read_img(self, path):\n    #     return cv2.imread(path, 0)\n   \n    def __getitem__(self, idx):   \n        window = 24\n        frame = self.frame[idx]\n        \n        if self.mode == 'train':\n            frame = frame + random.randint(-6, 6)\n\n        players = []\n        for p in self.players[idx]:\n            if p == 'G':\n                players.append(p)\n            else:\n                players.append(int(p))\n        \n        imgs = []\n        for view in ['Endzone', 'Sideline']:\n            video = self.game_play[idx] + f'_{view}.mp4'\n\n            tmp = video2helmets[video]\n#             tmp = tmp.query('@frame-@window<=frame<=@frame+@window')\n            tmp[tmp['frame'].between(frame-window, frame+window)]\n            tmp = tmp[tmp.nfl_player_id.isin(players)]#.sort_values(['nfl_player_id', 'frame'])\n            tmp_frames = tmp.frame.values\n            tmp = tmp.groupby('frame')[['left','width','top','height']].mean()\n#0.002s\n\n            bboxes = []\n            for f in range(frame-window, frame+window+1, 1):\n                if f in tmp_frames:\n                    x, w, y, h = tmp.loc[f][['left','width','top','height']]\n                    bboxes.append([x, w, y, h])\n                else:\n                    bboxes.append([np.nan, np.nan, np.nan, np.nan])\n            bboxes = pd.DataFrame(bboxes).interpolate(limit_direction='both').values\n            bboxes = bboxes[::4]\n\n            if bboxes.sum() > 0:\n                flag = 1\n            else:\n                flag = 0\n#0.03s\n                    \n            for i, f in enumerate(range(frame-window, frame+window+1, 4)):\n                img_new = np.zeros((256, 256), dtype=np.float32)\n\n                if flag == 1 and f <= video2frames[video]:\n                    img = cv2.imread(f'/kaggle/work/frames/{video}_{f:04d}.jpg', 0)\n\n                    x, w, y, h = bboxes[i]\n\n                    img = img[int(y+h/2)-128:int(y+h/2)+128,int(x+w/2)-128:int(x+w/2)+128].copy()\n                    img_new[:img.shape[0], :img.shape[1]] = img\n                    \n                imgs.append(img_new)\n#0.06s\n                \n        feature = np.float32(self.feature[idx])\n\n        img = np.array(imgs).transpose(1, 2, 0)    \n        img = self.aug(image=img)[\"image\"]\n        label = np.float32(self.df.contact.values[idx])\n\n        return img, feature, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:13.562794Z","iopub.execute_input":"2025-03-02T16:12:13.563325Z","iopub.status.idle":"2025-03-02T16:12:13.579598Z","shell.execute_reply.started":"2025-03-02T16:12:13.563287Z","shell.execute_reply":"2025-03-02T16:12:13.578863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img, feature, label = MyTestDataset(test_filtered, valid_aug, 'test')[0]\nplt.imshow(img.permute(1,2,0)[:,:,7])\nplt.show()\nimg.shape, feature, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:12:13.580669Z","iopub.execute_input":"2025-03-02T16:12:13.581000Z","iopub.status.idle":"2025-03-02T16:12:13.977098Z","shell.execute_reply.started":"2025-03-02T16:12:13.580974Z","shell.execute_reply":"2025-03-02T16:12:13.976261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_set = MyTestDataset(test_filtered, valid_aug, 'test')\ntest_loader = DataLoader(test_set, batch_size=CFG['valid_bs'], shuffle=False, num_workers=CFG['num_workers'], pin_memory=True)\n\nmodel = Model().to(device)\nmodel.load_state_dict(torch.load('/kaggle/input/resnet-position/weights_position_only.pth'))\n\nmodel.eval()\n    \ny_pred = []\nwith torch.no_grad():\n    tk = tqdm(test_loader, total=len(test_loader))\n    for step, batch in enumerate(tk):\n        img, feature, label = [x.to(device) for x in batch]\n        output = model(img, feature).squeeze(-1)\n\n        y_pred.extend(output.sigmoid().cpu().numpy())\n\ny_pred = np.array(y_pred)","metadata":{"execution":{"iopub.status.busy":"2025-03-02T16:12:54.841818Z","iopub.execute_input":"2025-03-02T16:12:54.842710Z","iopub.status.idle":"2025-03-02T16:33:14.923801Z","shell.execute_reply.started":"2025-03-02T16:12:54.842676Z","shell.execute_reply":"2025-03-02T16:33:14.922644Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"th = 0.29\n\ntest_filtered['contact'] = (y_pred >= th).astype('int')\n\nsub = pd.read_csv('/kaggle/input/nfl-player-contact-detection/sample_submission.csv')\n\nsub = sub.drop(\"contact\", axis=1).merge(test_filtered[['contact_id', 'contact']], how='left', on='contact_id')\nsub['contact'] = sub['contact'].fillna(0).astype('int')\n\nsub[[\"contact_id\", \"contact\"]].to_csv(\"submission.csv\", index=False)\n\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2025-03-02T16:33:14.926263Z","iopub.execute_input":"2025-03-02T16:33:14.927168Z","iopub.status.idle":"2025-03-02T16:33:15.078081Z","shell.execute_reply.started":"2025-03-02T16:33:14.927122Z","shell.execute_reply":"2025-03-02T16:33:15.077100Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}