{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":30886,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Nexar Dashcam Crash Prediction EDA","metadata":{}},{"cell_type":"markdown","source":"# Import library requirement","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport pandas.api.types\n\nimport sklearn.metrics\nimport os\nimport random\nimport glob\nimport cv2\nimport matplotlib.pyplot as plt\nfrom IPython.display import HTML, Video\nfrom base64 import b64encode\nimport re\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings('ignore', category=RuntimeWarning)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:11:08.388900Z","iopub.execute_input":"2025-06-24T09:11:08.389307Z","iopub.status.idle":"2025-06-24T09:11:08.394667Z","shell.execute_reply.started":"2025-06-24T09:11:08.389276Z","shell.execute_reply":"2025-06-24T09:11:08.393739Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compute Mean Average Precision (mAP)","metadata":{}},{"cell_type":"code","source":"# Competition Metric\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str, group_column_name: str = \"group\") -> float:\n    '''\n    Mean of the Average Precision AP. AP is calculated for each grouped by wrapping\n    https://scikit-learn.org/stable/modules/generated/sklearn.metrics.average_precision_score.html\n    and then the mean of the APs of all grouped is computed.\n\n    AP summarizes a precision-recall curve as the weighted mean of precisions\n    achieved at each threshold, with the increase in recall from the previous\n    threshold used as the weight:\n\n    .. math::\n    \\text{AP} = \\sum_n (R_n - R_{n-1}) P_n\n\n    where :math:`P_n` and :math:`R_n` are the precision and recall at the nth\n    threshold [1]_. This implementation is not interpolated and is different\n    from computing the area under the precision-recall curve with the\n    trapezoidal rule, which uses linear interpolation and can be too\n    optimistic.\n\n    Note: this implementation is restricted to the binary classification task.\n\n    Parameters\n    ----------\n    solution : ndarray of shape (n_samples,) or (n_samples, n_classes)\n    True binary labels or binary label indicators.\n\n    submission : ndarray of shape (n_samples,) or (n_samples, n_classes)\n    Target scores, can either be probability estimates of the positive\n    class, confidence values, or non-thresholded measure of decisions\n    (as returned by :term:`decision_function` on some classifiers).\n\n\n    Examples\n    --------\n\n    >>> import pandas as pd\n    >>> import numpy as np\n    >>> y_true = np.array([1, 0, 0, 0] + [1,0,0,1] + [1,0,1,1])\n    >>> y_true = pd.DataFrame(y_true)\n    >>> y_true[\"id\"] = range(len(y_true))\n    >>> y_true[\"group\"] = [\"a\", \"a\", \"a\", \"a\", \"b\", \"b\", \"b\", \"b\", \"c\", \"c\", \"c\", \"c\"]\n    >>> y_pred = np.array([0.1, 0.4, 0.35, 0.8] * 3)\n    >>> y_pred = pd.DataFrame(y_pred)\n    >>> y_pred[\"id\"] = range(len(y_pred))\n    >>> score(y_true.copy(), y_pred.copy(), \"id\", \"group\")\n    0.6018518518518519\n    '''\n\n    # Skip sorting and equality checks for the row_id_column since that should already be handled\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if not group_column_name in solution.columns:\n        raise ParticipantVisibleError('Missing group column in solution')\n\n    group = solution[group_column_name]\n    del solution[group_column_name]\n    groups = group.unique()\n\n    if not((len(submission.columns) == 1) or (len(submission.columns) == len(solution.columns))):\n        raise ParticipantVisibleError(f'Invalid number of submission columns. Found {len(submission.columns)}')\n\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        bad_dtypes = {x: submission[x].dtype  for x in submission.columns if not pandas.api.types.is_numeric_dtype(submission[x])}\n        raise ParticipantVisibleError(f'Invalid submission data types found: {bad_dtypes}')\n\n    if submission.max().max() > 1 or submission.min().min() < 0:\n        raise ParticipantVisibleError('Submitted values were not valid probabilities')\n\n    solution = solution.values\n    submission = submission.values\n\n    score_result = np.mean([\n        sklearn.metrics.average_precision_score(solution[group == g], submission[group == g])\n        for g in groups\n    ])\n\n    return score_result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.364517Z","iopub.execute_input":"2025-06-24T08:13:51.365000Z","iopub.status.idle":"2025-06-24T08:13:51.373727Z","shell.execute_reply.started":"2025-06-24T08:13:51.364973Z","shell.execute_reply":"2025-06-24T08:13:51.372713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\ntest = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\n\nss = pd.read_csv('/kaggle/input/nexar-collision-prediction/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.375845Z","iopub.execute_input":"2025-06-24T08:13:51.376110Z","iopub.status.idle":"2025-06-24T08:13:51.431336Z","shell.execute_reply.started":"2025-06-24T08:13:51.376088Z","shell.execute_reply":"2025-06-24T08:13:51.430409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.sort_values(by='id')\ntest = test.sort_values(by='id')\nss = ss.sort_values(by='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.432751Z","iopub.execute_input":"2025-06-24T08:13:51.433006Z","iopub.status.idle":"2025-06-24T08:13:51.462369Z","shell.execute_reply.started":"2025-06-24T08:13:51.432984Z","shell.execute_reply":"2025-06-24T08:13:51.461435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.463393Z","iopub.execute_input":"2025-06-24T08:13:51.463767Z","iopub.status.idle":"2025-06-24T08:13:51.490749Z","shell.execute_reply.started":"2025-06-24T08:13:51.463725Z","shell.execute_reply":"2025-06-24T08:13:51.489492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.491798Z","iopub.execute_input":"2025-06-24T08:13:51.492137Z","iopub.status.idle":"2025-06-24T08:13:51.500875Z","shell.execute_reply.started":"2025-06-24T08:13:51.492097Z","shell.execute_reply":"2025-06-24T08:13:51.499864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ss.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.501793Z","iopub.execute_input":"2025-06-24T08:13:51.502159Z","iopub.status.idle":"2025-06-24T08:13:51.520622Z","shell.execute_reply.started":"2025-06-24T08:13:51.502133Z","shell.execute_reply":"2025-06-24T08:13:51.519577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read the image locations\ntrain_filenames = glob.glob('/kaggle/input/nexar-collision-prediction/train/*.mp4')\ntest_filenames = glob.glob('/kaggle/input/nexar-collision-prediction/test/*.mp4')\n\n# Sort by id\ntrain_filenames = sorted(train_filenames)\ntest_filenames = sorted(test_filenames)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:13:51.523562Z","iopub.execute_input":"2025-06-24T08:13:51.523857Z","iopub.status.idle":"2025-06-24T08:13:51.597789Z","shell.execute_reply.started":"2025-06-24T08:13:51.523823Z","shell.execute_reply":"2025-06-24T08:13:51.597037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm trích xuấxuất ID từ tên file \n#video_path = '/kaggle/input/nexar-collision-prediction/train/00000.mp4'\ndef get_id(video_path):\n    match = re.search(r'(\\d+)\\.mp4$', video_path)\n    return match.group(1) if match else None\nget_id(video_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:58:15.941792Z","iopub.execute_input":"2025-06-24T08:58:15.942182Z","iopub.status.idle":"2025-06-24T08:58:15.948348Z","shell.execute_reply.started":"2025-06-24T08:58:15.942150Z","shell.execute_reply":"2025-06-24T08:58:15.947530Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Video Metadata \n\nConstant Features\n- height, width\n\nVarying Features \n- fps - is not the same though\n","metadata":{}},{"cell_type":"code","source":"WIDTH = 1280\nHEIGHT = 720","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:58:19.527207Z","iopub.execute_input":"2025-06-24T08:58:19.527558Z","iopub.status.idle":"2025-06-24T08:58:19.531967Z","shell.execute_reply.started":"2025-06-24T08:58:19.527531Z","shell.execute_reply":"2025-06-24T08:58:19.530847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_metadata(video_paths):\n    fps_values = []\n    frame_counts = []\n    total_durations = []\n    for video_path in video_paths:\n        cap = cv2.VideoCapture(video_path)\n        if not cap.isOpened():\n            print('Error: Cannot open video file.')\n            exit()\n        fps = cap.get(cv2.CAP_PROP_FPS)\n        frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        total_duration = frame_count / fps if fps > 0 else 0\n        \n        fps_values.append(fps)\n        frame_counts.append(frame_count)\n        total_durations.append(total_duration)\n        cap.release()\n\n    results = {\n        'fps': fps_values,\n        'frame_count': frame_counts,\n        'total_duration': total_durations\n        }\n    \n    return results\n\n\n# Thêm metadata vào train/test (nếu chưa có)\nif 'fps' not in train.columns:\n    train_metadata = get_metadata(train_filenames)\n    for key in train_metadata:\n        train[key] = train_metadata[key]\nif 'fps' not in test.columns:\n    test_metadata = get_metadata(test_filenames)\n    for key in test_metadata:\n        test[key] = test_metadata[key]\n\n# --- Phân tích metadata ---\nprint(\"--- Các cột trong train.csv ---\")\nprint(train.info())\nprint(\"\\n--- Mẫu dữ liệu đầu tiên ---\")\nprint(train.head())\nprint(\"\\n--- Thống kê mô tả ---\")\nprint(train.describe())\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:02:44.786455Z","iopub.execute_input":"2025-06-24T09:02:44.786812Z","iopub.status.idle":"2025-06-24T09:02:44.824969Z","shell.execute_reply.started":"2025-06-24T09:02:44.786782Z","shell.execute_reply":"2025-06-24T09:02:44.824117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"--- Các cột trong train.csv ---\")\nprint(train.info())\nprint(\"\\n--- Mẫu dữ liệu đầu tiên ---\")\nprint(train.head())\n# Kiểm tra các cột\nprint(\"\\nColumns:\", train.columns.tolist())\n# Nếu có cột tương tự 'label', tìm tên chính xác","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:03:37.282911Z","iopub.execute_input":"2025-06-24T09:03:37.283274Z","iopub.status.idle":"2025-06-24T09:03:37.299107Z","shell.execute_reply.started":"2025-06-24T09:03:37.283247Z","shell.execute_reply":"2025-06-24T09:03:37.298251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Phân phối nhãn (target)\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train, x='target')\nplt.title('Phân phối nhãn (0: Không va chạm, 1: Va chạm)')\nplt.xlabel('Target')\nplt.ylabel('Số lượng')\nplt.show()\n# Kiểm tra sự mất cân bằng\nlabel_counts = train['target'].value_counts()\nprint(\"\\n--- Tỷ lệ nhãn ---\")\nprint(f\"Không va chạm (0): {label_counts[0]} ({label_counts[0]/len(train)*100:.2f}%)\")\nprint(f\"Va chạm (1): {label_counts[1]} ({label_counts[1]/len(train)*100:.2f}%)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:04:12.187088Z","iopub.execute_input":"2025-06-24T09:04:12.187410Z","iopub.status.idle":"2025-06-24T09:04:12.371257Z","shell.execute_reply.started":"2025-06-24T09:04:12.187385Z","shell.execute_reply":"2025-06-24T09:04:12.370340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Phân tích thời lượng video\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train, x='total_duration', bins=30)\nplt.title('Phân phối thời lượng video (giây)')\nplt.xlabel('Thời lượng (s)')\nplt.ylabel('Số lượng')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:04:46.233920Z","iopub.execute_input":"2025-06-24T09:04:46.234306Z","iopub.status.idle":"2025-06-24T09:04:46.506046Z","shell.execute_reply.started":"2025-06-24T09:04:46.234278Z","shell.execute_reply":"2025-06-24T09:04:46.505128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Phân tích số frame\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train, x='frame_count', bins=30)\nplt.title('Phân phối số frame')\nplt.xlabel('Số frame')\nplt.ylabel('Số lượng')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:05:02.796555Z","iopub.execute_input":"2025-06-24T09:05:02.796887Z","iopub.status.idle":"2025-06-24T09:05:03.078587Z","shell.execute_reply.started":"2025-06-24T09:05:02.796861Z","shell.execute_reply":"2025-06-24T09:05:03.077584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. Phân tích FPS\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train, x='fps', bins=10)\nplt.title('Phân phối FPS')\nplt.xlabel('FPS')\nplt.ylabel('Số lượng')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:05:21.318565Z","iopub.execute_input":"2025-06-24T09:05:21.318953Z","iopub.status.idle":"2025-06-24T09:05:21.558939Z","shell.execute_reply.started":"2025-06-24T09:05:21.318920Z","shell.execute_reply":"2025-06-24T09:05:21.558144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. Phân tích thời gian sự kiện (time_of_event)\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train[train['target'] == 1], x='time_of_event', bins=30, label='Va chạm', color='red', alpha=0.5)\nsns.histplot(data=train[train['target'] == 0], x='time_of_event', bins=30, label='Không va chạm', color='blue', alpha=0.5)\nplt.title('Phân phối thời gian sự kiện (time_of_event)')\nplt.xlabel('Thời gian (s)')\nplt.ylabel('Số lượng')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:05:45.421310Z","iopub.execute_input":"2025-06-24T09:05:45.421620Z","iopub.status.idle":"2025-06-24T09:05:45.709776Z","shell.execute_reply.started":"2025-06-24T09:05:45.421596Z","shell.execute_reply":"2025-06-24T09:05:45.708801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 7. Phân tích thời gian cảnh báo (time_of_alert)\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train[train['target'] == 1], x='time_of_alert', bins=30, label='Va chạm', color='red', alpha=0.5)\nsns.histplot(data=train[train['target'] == 0], x='time_of_alert', bins=30, label='Không va chạm', color='blue', alpha=0.5)\nplt.title('Phân phối thời gian cảnh báo (time_of_alert)')\nplt.xlabel('Thời gian (s)')\nplt.ylabel('Số lượng')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:06:16.715999Z","iopub.execute_input":"2025-06-24T09:06:16.716385Z","iopub.status.idle":"2025-06-24T09:06:17.078192Z","shell.execute_reply.started":"2025-06-24T09:06:16.716354Z","shell.execute_reply":"2025-06-24T09:06:17.077121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 8. So sánh time_of_event và time_of_alert\ntrain['alert_event_diff'] = train['time_of_alert'] - train['time_of_event']\nplt.figure(figsize=(10, 6))\nsns.histplot(data=train[train['target'] == 1], x='alert_event_diff', bins=30, color='green')\nplt.title('Phân phối chênh lệch thời gian (time_of_alert - time_of_event) cho va chạm')\nplt.xlabel('Chênh lệch thời gian (s)')\nplt.ylabel('Số lượng')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:06:48.049159Z","iopub.execute_input":"2025-06-24T09:06:48.049492Z","iopub.status.idle":"2025-06-24T09:06:48.339702Z","shell.execute_reply.started":"2025-06-24T09:06:48.049467Z","shell.execute_reply":"2025-06-24T09:06:48.338660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The NEW main training directory of the dataset after moving\ndataset_train_dir = \"/kaggle/input/nexar-collision-prediction/train\"\n\n# Check if the target directory exists\nif not os.path.isdir(dataset_train_dir):\n    print(f\"Error: Directory not found: {dataset_train_dir}\")\n    print(f\"Please check if the '{target_subfolder}' folder exists inside '{dataset_train_dir}'\")\nelse:\n    # List all files in the target video directory\n    files = os.listdir(dataset_train_dir)\n\n    # Filter for video files based on common extensions\n    video_files = [f for f in files if f.lower().endswith(('.mp4', '.avi', '.mkv', '.mov', '.webm'))]\n\n    if not video_files:\n        print(f\"No video files found in the directory: {dataset_train_dir}\")\n    else:\n        print(f\"Found {len(video_files)} video files in {dataset_train_dir}:\")\n        # for i, video_file in enumerate(video_files):\n        #     print(f\"{i+1}. {video_file}\")\n\n        # --- Displaying a random video using HTML embedding ---\n\n        # Check if there's at least one video file\n        if video_files:\n            # Choose a random video from the list\n            chosen_video_name = random.choice(video_files)\n            video_file_path = os.path.join(dataset_train_dir, chosen_video_name)\n\n            print(f\"\\nDisplaying a random video using HTML embedding: {chosen_video_name}\")\n\n            try:\n                # Read the video file in binary mode\n                with open(video_file_path, 'rb') as f:\n                    mp4 = f.read()\n\n                # Encode the video data in base64\n                data_url = \"data:video/mp4;base64,\" + b64encode(mp4).decode()\n\n                # Create and display the HTML video player\n                display(HTML(\"\"\"\n                <video width=400 controls>\n                      <source src=\"%s\" type=\"video/mp4\">\n                </video>\n                \"\"\" % data_url))\n\n            except FileNotFoundError:\n                print(f\"Error: Video file not found at: {video_file_path}\")\n            except Exception as e:\n                print(f\"An error occurred during HTML embedding: {e}\")\n        else:\n            print(\"No video files available to display.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:08:02.368963Z","iopub.execute_input":"2025-06-24T09:08:02.369350Z","iopub.status.idle":"2025-06-24T09:08:03.962208Z","shell.execute_reply.started":"2025-06-24T09:08:02.369323Z","shell.execute_reply":"2025-06-24T09:08:03.961183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm hiển thị frame và tạo video\ndef display_frames_and_video(video_path, num_frames=5, time_of_event=None, output_video_path='output_video.mp4'):\n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f'Error: Cannot open {video_path}')\n        return\n    \n    frames = []\n    frame_count = 0\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    frame_width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\n    frame_height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n    \n    # Khởi tạo video writer\n    fourcc = cv2.VideoWriter_fourcc(*'mp4v')\n    out = cv2.VideoWriter(output_video_path, fourcc, fps, (frame_width, frame_height))\n    \n    # Di chuyển đến thời điểm time_of_event (nếu có)\n    if time_of_event is not None and fps > 0:\n        cap.set(cv2.CAP_PROP_POS_MSEC, time_of_event * 1000)\n        frame_count = int(time_of_event * fps)\n    \n    # Đọc và lưu frame\n    target_frame_count = int(time_of_event * fps) + num_frames if time_of_event else num_frames\n    while frame_count < target_frame_count:\n        ret, frame = cap.read()\n        if not ret:\n            break\n        # Lưu frame để hiển thị tĩnh\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        frames.append(frame_rgb)\n        # Viết frame vào video\n        out.write(frame)\n        frame_count += 1\n    \n    # Giải phóng tài nguyên\n    cap.release()\n    out.release()\n    \n    # Hiển thị các frame tĩnh\n    plt.figure(figsize=(15, 3))\n    for i, frame in enumerate(frames):\n        plt.subplot(1, min(num_frames, len(frames)), i+1)\n        plt.imshow(frame)\n        plt.axis('off')\n        plt.title(f'Frame {i+1}')\n    plt.show()\n    \n    # Hiển thị video\n    print(f\"Phát đoạn video từ {video_path}\")\n    display(Video(output_video_path, embed=True, width=600))\n\n# Hiển thị frame và video từ video có va chạm và không va chạm\ncollision_video = train[train['target'] == 1]['id'].iloc[0]\nno_collision_video = train[train['target'] == 0]['id'].iloc[0]\ncollision_path = [f for f in train_filenames if get_id(f) == str(collision_video).zfill(5)][0]\nno_collision_path = [f for f in train_filenames if get_id(f) == str(no_collision_video).zfill(5)][0]\n\nprint(f\"\\nHiển thị video có va chạm (ID: {collision_video})\")\ndisplay_frames_and_video(collision_path, time_of_event=train[train['id'] == collision_video]['time_of_event'].iloc[0], output_video_path=f'collision_{collision_video}.mp4')\n\nprint(f\"\\nHiển thị video không va chạm (ID: {no_collision_video})\")\ndisplay_frames_and_video(no_collision_path, output_video_path=f'no_collision_{no_collision_video}.mp4')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:11:42.453973Z","iopub.execute_input":"2025-06-24T09:11:42.454383Z","iopub.status.idle":"2025-06-24T09:11:44.761187Z","shell.execute_reply.started":"2025-06-24T09:11:42.454355Z","shell.execute_reply":"2025-06-24T09:11:44.760164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Phân tích video ---\ndef display_frames(video_path, num_frames=5, time_of_event=None):\n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f'Error: Cannot open {video_path}')\n        return\n    frames = []\n    frame_count = 0\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    if time_of_event is not None and fps > 0:\n        # Di chuyển đến frame gần time_of_event\n        cap.set(cv2.CAP_PROP_POS_MSEC, time_of_event * 1000)\n        frame_count = int(time_of_event * fps)\n    while frame_count < (int(time_of_event * fps) + num_frames if time_of_event else num_frames):\n        ret, frame = ret, frame = cap.read()\n        if not ret:\n            break\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        frames.append(frame)\n        frame_count += 1\n    cap.release()\n    plt.figure(figsize=(15, 3))\n    for i, frame in enumerate(frames):\n        plt.subplot(1, min(num_frames, len(frames)), i+1)\n        plt.imshow(frame)\n        plt.axis('off')\n        plt.title(f'Frame {i+1}')\n    plt.show()\n\n# Hiển thị frame từ video có va chạm và không va chạm\ncollision_video = train[train['target'] == 1]['id'].iloc[0]\nno_collision_video = train[train['target'] == 0]['id'].iloc[0]\ncollision_path = [f for f in train_filenames if get_id(f) == str(collision_video).zfill(5)][0]\nno_collision_path = [f for f in train_filenames if get_id(f) == str(no_collision_video).zfill(5)][0]\n\nprint(f\"\\nHiển thị video có va chạm (ID: {collision_video})\")\ndisplay_frames(collision_path, time_of_event=train[train['id'] == collision_video]['time_of_event'].iloc[0])\nprint(f\"\\nHiển thị video không va chạm (ID: {no_collision_video})\")\ndisplay_frames(no_collision_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T09:07:43.847431Z","iopub.execute_input":"2025-06-24T09:07:43.847793Z","iopub.status.idle":"2025-06-24T09:07:45.986681Z","shell.execute_reply.started":"2025-06-24T09:07:43.847766Z","shell.execute_reply":"2025-06-24T09:07:45.985659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.to_csv('train.csv', index=False)\ntest.to_csv('test.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T08:15:03.595363Z","iopub.execute_input":"2025-06-24T08:15:03.595924Z","iopub.status.idle":"2025-06-24T08:15:03.619184Z","shell.execute_reply.started":"2025-06-24T08:15:03.595877Z","shell.execute_reply":"2025-06-24T08:15:03.618132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}}]}