{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.16","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"},{"sourceId":10807353,"sourceType":"datasetVersion","datasetId":6708458},{"sourceId":10808094,"sourceType":"datasetVersion","datasetId":6709031}],"dockerImageVersionId":30887,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         (os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-20T20:52:59.037356Z","iopub.execute_input":"2025-02-20T20:52:59.037690Z","iopub.status.idle":"2025-02-20T20:52:59.041596Z","shell.execute_reply.started":"2025-02-20T20:52:59.037666Z","shell.execute_reply":"2025-02-20T20:52:59.040637Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport torchvision.transforms as transforms\nimport numpy as np\nimport cv2\nimport os\nfrom tqdm import tqdm\nimport pandas as pd\n\n# -------------------------------\n# Device Setup for CUDA\n# -------------------------------\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint(\"Using device:\", device)\n\n# -------------------------------\n# Load and Modify ResNet-18\n# -------------------------------\nresnet18 = models.resnet18(pretrained=True)\nresnet18 = nn.Sequential(*list(resnet18.children())[:-1])  # Remove final FC layer\nresnet18 = resnet18.to(device).eval()  # Set to evaluation mode\n\nresnet_transform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225])\n])\n\nFPS_TARGET = 5        # Set your target FPS for sampling (adjust if needed)\nSEQUENCE_LENGTH = 5     # Number of frames per sequence\n\n# -------------------------------\n# Per-Frame Label Computation Function\n# -------------------------------\ndef compute_frame_label(t, event_time, sigma_before=1.0, sigma_after=1.0):\n    \"\"\"\n    Compute a soft target label for a single frame timestamp 't' given the event_time.\n    Returns 1.0 if 't' is (approximately) equal to event_time, and otherwise a value in (0,1)\n    based on an asymmetric Gaussian decay.\n    \"\"\"\n    # Use a small tolerance (e.g., 0.033 sec ~ one frame at 30 FPS) to consider t as event frame.\n    if np.isclose(t, event_time, atol=0.18):\n        return 1.0\n    if t < event_time:\n        # Before event: decay from 1 as the time difference increases.\n        return np.exp(-((event_time - t)**2) / (2 * sigma_before**2))\n    else:\n        # After event: decay from 1 as well.\n        return np.exp(-((t - event_time)**2) / (2 * sigma_after**2))\n\n# -------------------------------\n# Feature Extraction & Sequence Generation\n# -------------------------------\ndef extract_features_and_labels(video_dir, df):\n    \"\"\"\n    For each video in the provided dataframe, extract one sequence of feature vectors using the pretrained ResNet-18.\n    For positive cases (where time_of_event is provided), select a sequence containing that event \n    and compute a per-frame soft target vector based on time_of_event severity.\n    For negative cases, select a sequence from the middle and assign a target vector of all zeros.\n    \n    Returns:\n      video_sequences: numpy array of shape (num_videos, SEQUENCE_LENGTH, feature_dim)\n      video_labels: numpy array of shape (num_videos, SEQUENCE_LENGTH)\n    \"\"\"\n    video_sequences = []\n    video_labels = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        video_id = row[\"id\"]\n        video_path = os.path.join(video_dir, f\"{video_id}.mp4\")\n        if not os.path.exists(video_path):\n            continue\n        \n        # For positive cases, time_of_event is provided; for negatives it will be NaN.\n        event_time = row[\"time_of_event\"]\n        if pd.isna(event_time):\n            event_time = -1  # Mark negative cases\n        \n        cap = cv2.VideoCapture(video_path)\n        fps = cap.get(cv2.CAP_PROP_FPS)\n        total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        if fps == 0 or np.isnan(fps):\n            fps = 30\n        \n        frame_step = int(fps / FPS_TARGET)\n        sampled_features = []\n        sampled_timestamps = []\n        \n        current_frame = 0\n        while current_frame < total_frames:\n            cap.set(cv2.CAP_PROP_POS_FRAMES, current_frame)\n            ret, frame = cap.read()\n            if not ret:\n                break\n            # Extract feature using ResNet-18.\n            frame_tensor = resnet_transform(frame).unsqueeze(0).to(device)\n            with torch.no_grad():\n                feature = resnet18(frame_tensor)\n            feature = feature.squeeze().cpu().numpy()  # e.g., shape (512,)\n            timestamp = current_frame / fps\n            sampled_features.append(feature)\n            sampled_timestamps.append(timestamp)\n            current_frame += frame_step\n        cap.release()\n        \n        sampled_features = np.array(sampled_features)\n        sampled_timestamps = np.array(sampled_timestamps)\n        \n        if len(sampled_features) < SEQUENCE_LENGTH:\n            continue\n        \n        # Positive case: event_time provided (target == 1 in CSV)\n        if event_time != -1:\n            event_idx = np.argmin(np.abs(sampled_timestamps - event_time))\n            start_idx = max(0, event_idx - 2)\n            end_idx = start_idx + SEQUENCE_LENGTH\n            if end_idx > len(sampled_features):\n                end_idx = len(sampled_features)\n                start_idx = end_idx - SEQUENCE_LENGTH\n            sequence = sampled_features[start_idx:end_idx]\n            sequence_times = sampled_timestamps[start_idx:end_idx]\n            # Compute a per-frame target label\n            labels_seq = [compute_frame_label(t, event_time, sigma_before=1.0, sigma_after=1.0)\n                          for t in sequence_times]\n        else:\n            # Negative case: assign a sequence from the middle with target 0 for every frame.\n            mid_idx = len(sampled_features) // 2\n            start_idx = max(0, mid_idx - SEQUENCE_LENGTH//2)\n            end_idx = start_idx + SEQUENCE_LENGTH\n            if end_idx > len(sampled_features):\n                end_idx = len(sampled_features)\n                start_idx = end_idx - SEQUENCE_LENGTH\n            sequence = sampled_features[start_idx:end_idx]\n            labels_seq = [0.0] * SEQUENCE_LENGTH\n        \n        video_sequences.append(sequence)\n        video_labels.append(labels_seq)\n    \n    return np.array(video_sequences), np.array(video_labels)\n\n# -------------------------------\n# Create Dataset and DataLoader\n# -------------------------------\nclass AccidentFeatureDataset(torch.utils.data.Dataset):\n    def __init__(self, features, labels):\n        self.features = features  # shape: (num_samples, SEQUENCE_LENGTH, feature_dim)\n        self.labels = labels      # shape: (num_samples, SEQUENCE_LENGTH)\n    \n    def __len__(self):\n        return len(self.labels)\n    \n    def __getitem__(self, idx):\n        sequence = torch.tensor(self.features[idx], dtype=torch.float32)\n        label_seq = torch.tensor(self.labels[idx], dtype=torch.float32)\n        return sequence, label_seq\n\n# -------------------------------\n# Main Processing Pipeline\n# -------------------------------\n# Load metadata CSV.\ndf = pd.read_csv(\"/kaggle/input/nexar-collision-prediction/train.csv\")\n# Ensure video ids are zero-padded (e.g., \"00001\").\ndf[\"id\"] = df[\"id\"].apply(lambda x: str(x).zfill(5))\n\n# Limit processing to the first 500 videos.\ndf_subset = df.head(500)\n\n# Set the video directory.\nTRAIN_VIDEO_DIR = \"/kaggle/input/nexar-collision-prediction/train\"\n\n# Extract features and per-frame labels for the subset.\nsequences, labels = extract_features_and_labels(TRAIN_VIDEO_DIR, df_subset)\n\nprint(\"Extracted sequences shape:\", sequences.shape)\nprint(\"Extracted labels shape:\", labels.shape)\nnp.save(\"train_features.npy\", sequences)\nnp.save(\"train_labels.npy\", labels)\nprint(\"Features and labels saved successfully!\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}