{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":30886,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Nexar Dashcam Crash Prediction EDA","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport pandas.api.types\n\nimport sklearn.metrics\n\n\nimport glob\nimport cv2\nimport matplotlib.pyplot as plt\n\nimport re\n\n\nimport warnings\nwarnings.filterwarnings('ignore', category=RuntimeWarning)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:33:37.648824Z","iopub.execute_input":"2025-02-16T20:33:37.649162Z","iopub.status.idle":"2025-02-16T20:33:40.672514Z","shell.execute_reply.started":"2025-02-16T20:33:37.649118Z","shell.execute_reply":"2025-02-16T20:33:40.671333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Competition Metric\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str, group_column_name: str = \"group\") -> float:\n    '''\n    Mean of the Average Precision AP. AP is calculated for each grouped by wrapping\n    https://scikit-learn.org/stable/modules/generated/sklearn.metrics.average_precision_score.html\n    and then the mean of the APs of all grouped is computed.\n\n    AP summarizes a precision-recall curve as the weighted mean of precisions\n    achieved at each threshold, with the increase in recall from the previous\n    threshold used as the weight:\n\n    .. math::\n    \\text{AP} = \\sum_n (R_n - R_{n-1}) P_n\n\n    where :math:`P_n` and :math:`R_n` are the precision and recall at the nth\n    threshold [1]_. This implementation is not interpolated and is different\n    from computing the area under the precision-recall curve with the\n    trapezoidal rule, which uses linear interpolation and can be too\n    optimistic.\n\n    Note: this implementation is restricted to the binary classification task.\n\n    Parameters\n    ----------\n    solution : ndarray of shape (n_samples,) or (n_samples, n_classes)\n    True binary labels or binary label indicators.\n\n    submission : ndarray of shape (n_samples,) or (n_samples, n_classes)\n    Target scores, can either be probability estimates of the positive\n    class, confidence values, or non-thresholded measure of decisions\n    (as returned by :term:`decision_function` on some classifiers).\n\n\n    Examples\n    --------\n\n    >>> import pandas as pd\n    >>> import numpy as np\n    >>> y_true = np.array([1, 0, 0, 0] + [1,0,0,1] + [1,0,1,1])\n    >>> y_true = pd.DataFrame(y_true)\n    >>> y_true[\"id\"] = range(len(y_true))\n    >>> y_true[\"group\"] = [\"a\", \"a\", \"a\", \"a\", \"b\", \"b\", \"b\", \"b\", \"c\", \"c\", \"c\", \"c\"]\n    >>> y_pred = np.array([0.1, 0.4, 0.35, 0.8] * 3)\n    >>> y_pred = pd.DataFrame(y_pred)\n    >>> y_pred[\"id\"] = range(len(y_pred))\n    >>> score(y_true.copy(), y_pred.copy(), \"id\", \"group\")\n    0.6018518518518519\n    '''\n\n    # Skip sorting and equality checks for the row_id_column since that should already be handled\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if not group_column_name in solution.columns:\n        raise ParticipantVisibleError('Missing group column in solution')\n\n    group = solution[group_column_name]\n    del solution[group_column_name]\n    groups = group.unique()\n\n    if not((len(submission.columns) == 1) or (len(submission.columns) == len(solution.columns))):\n        raise ParticipantVisibleError(f'Invalid number of submission columns. Found {len(submission.columns)}')\n\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        bad_dtypes = {x: submission[x].dtype  for x in submission.columns if not pandas.api.types.is_numeric_dtype(submission[x])}\n        raise ParticipantVisibleError(f'Invalid submission data types found: {bad_dtypes}')\n\n    if submission.max().max() > 1 or submission.min().min() < 0:\n        raise ParticipantVisibleError('Submitted values were not valid probabilities')\n\n    solution = solution.values\n    submission = submission.values\n\n    score_result = np.mean([\n        sklearn.metrics.average_precision_score(solution[group == g], submission[group == g])\n        for g in groups\n    ])\n\n    return score_result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:33:40.673739Z","iopub.execute_input":"2025-02-16T20:33:40.674287Z","iopub.status.idle":"2025-02-16T20:33:40.684550Z","shell.execute_reply.started":"2025-02-16T20:33:40.674253Z","shell.execute_reply":"2025-02-16T20:33:40.683201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\ntest = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\n\nss = pd.read_csv('/kaggle/input/nexar-collision-prediction/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:33:40.685802Z","iopub.execute_input":"2025-02-16T20:33:40.686311Z","iopub.status.idle":"2025-02-16T20:33:40.756446Z","shell.execute_reply.started":"2025-02-16T20:33:40.686252Z","shell.execute_reply":"2025-02-16T20:33:40.755319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.sort_values(by='id')\ntest = test.sort_values(by='id')\nss = ss.sort_values(by='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:50:59.208936Z","iopub.execute_input":"2025-02-16T20:50:59.209450Z","iopub.status.idle":"2025-02-16T20:50:59.217601Z","shell.execute_reply.started":"2025-02-16T20:50:59.209416Z","shell.execute_reply":"2025-02-16T20:50:59.216343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T21:11:52.365516Z","iopub.execute_input":"2025-02-16T21:11:52.365872Z","iopub.status.idle":"2025-02-16T21:11:52.378594Z","shell.execute_reply.started":"2025-02-16T21:11:52.365841Z","shell.execute_reply":"2025-02-16T21:11:52.377496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:51:37.144976Z","iopub.execute_input":"2025-02-16T20:51:37.145393Z","iopub.status.idle":"2025-02-16T20:51:37.155004Z","shell.execute_reply.started":"2025-02-16T20:51:37.145361Z","shell.execute_reply":"2025-02-16T20:51:37.153808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ss.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:51:03.734341Z","iopub.execute_input":"2025-02-16T20:51:03.734673Z","iopub.status.idle":"2025-02-16T20:51:03.743512Z","shell.execute_reply.started":"2025-02-16T20:51:03.734647Z","shell.execute_reply":"2025-02-16T20:51:03.742458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read the image locations\ntrain_filenames = glob.glob('/kaggle/input/nexar-collision-prediction/train/*.mp4')\ntest_filenames = glob.glob('/kaggle/input/nexar-collision-prediction/test/*.mp4')\n\n# Sort by id\ntrain_filenames = sorted(train_filenames)\ntest_filenames = sorted(test_filenames)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:51:12.119014Z","iopub.execute_input":"2025-02-16T20:51:12.119400Z","iopub.status.idle":"2025-02-16T20:51:12.134738Z","shell.execute_reply.started":"2025-02-16T20:51:12.119364Z","shell.execute_reply":"2025-02-16T20:51:12.133357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# video_path = '/kaggle/input/nexar-collision-prediction/train/00000.mp4'\n\n# def get_id(video_path):\n#     match = re.search(r'(\\d+)\\.mp4$', video_path)\n#     return match.group(1) if match else None\n\n# get_id(video_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:51:53.084768Z","iopub.execute_input":"2025-02-16T20:51:53.085227Z","iopub.status.idle":"2025-02-16T20:51:53.092931Z","shell.execute_reply.started":"2025-02-16T20:51:53.085190Z","shell.execute_reply":"2025-02-16T20:51:53.091436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Video Metadata \n\nConstant Features\n- height, width\n\nVarying Features \n- fps - is not the same though\n","metadata":{}},{"cell_type":"code","source":"WIDTH = 1280\nHEIGHT = 720","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:56:42.662518Z","iopub.execute_input":"2025-02-16T20:56:42.662909Z","iopub.status.idle":"2025-02-16T20:56:42.667607Z","shell.execute_reply.started":"2025-02-16T20:56:42.662880Z","shell.execute_reply":"2025-02-16T20:56:42.666280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_metadata(video_paths):\n    fps_values = []\n    frame_counts = []\n    total_durations = []\n    for video_path in video_paths:\n        cap = cv2.VideoCapture(video_path)\n        if not cap.isOpened():\n            print('Error: Cannot open video file.')\n            exit()\n        fps = cap.get(cv2.CAP_PROP_FPS)\n        frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        total_duration = frame_count / fps if fps > 0 else 0\n        \n        fps_values.append(fps)\n        frame_counts.append(frame_count)\n        total_durations.append(total_duration)\n        cap.release()\n\n    results = {\n        'fps': fps_values,\n        'frame_count': frame_counts,\n        'total_duration': total_durations\n        }\n    \n    return results\n\n\ntrain_metadata = get_metadata(train_filenames)\nfor key in train_metadata:\n    train[key] = train_metadata[key]\n\ntest_metadata = get_metadata(test_filenames)\nfor key in test_metadata:\n    test[key] = test_metadata[key]\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T21:08:13.450563Z","iopub.execute_input":"2025-02-16T21:08:13.450886Z","iopub.status.idle":"2025-02-16T21:08:50.612651Z","shell.execute_reply.started":"2025-02-16T21:08:13.450860Z","shell.execute_reply":"2025-02-16T21:08:50.611311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load video\nvideo_path = train_filenames[0]  \ncap = cv2.VideoCapture(video_path)\n\nif not cap.isOpened():\n    print('Error: Cannot open video file.')\n    exit()\n\nframe_count = 0\n\n# Read and process frames\nwhile frame_count < 5:  \n    ret, frame = cap.read()\n    if not ret:\n        break\n    \n    # Convert to grayscale\n    gray_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n    \n    # Display frame using matplotlib\n    plt.imshow(gray_frame, cmap='gray')\n    plt.axis('off')  # Hide axes\n    plt.show()\n    \n    frame_count += 1\n\ncap.release()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T20:33:41.023083Z","iopub.execute_input":"2025-02-16T20:33:41.023383Z","iopub.status.idle":"2025-02-16T20:33:42.293440Z","shell.execute_reply.started":"2025-02-16T20:33:41.023358Z","shell.execute_reply":"2025-02-16T20:33:42.292182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.to_csv('train.csv', index=False)\ntest.to_csv('test.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T21:15:10.419685Z","iopub.execute_input":"2025-02-16T21:15:10.420066Z","iopub.status.idle":"2025-02-16T21:15:10.444667Z","shell.execute_reply.started":"2025-02-16T21:15:10.420014Z","shell.execute_reply":"2025-02-16T21:15:10.443225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explore the Competition Metric","metadata":{}}]}