{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":30887,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\ndata_path = \"/kaggle/input/nexar-collision-prediction/\"\nprint(os.listdir(data_path))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:13.156201Z","iopub.execute_input":"2025-02-19T17:24:13.156606Z","iopub.status.idle":"2025-02-19T17:24:13.163370Z","shell.execute_reply.started":"2025-02-19T17:24:13.156570Z","shell.execute_reply":"2025-02-19T17:24:13.162482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:17.686520Z","iopub.execute_input":"2025-02-19T17:24:17.686891Z","iopub.status.idle":"2025-02-19T17:24:17.691804Z","shell.execute_reply.started":"2025-02-19T17:24:17.686862Z","shell.execute_reply":"2025-02-19T17:24:17.690909Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Explore the Dataset\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load train.csv\ntrain_df = pd.read_csv(f\"{data_path}/train.csv\")\n\nprint(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:19.316509Z","iopub.execute_input":"2025-02-19T17:24:19.316845Z","iopub.status.idle":"2025-02-19T17:24:19.672140Z","shell.execute_reply.started":"2025-02-19T17:24:19.316819Z","shell.execute_reply":"2025-02-19T17:24:19.671230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:26.480539Z","iopub.execute_input":"2025-02-19T17:24:26.480870Z","iopub.status.idle":"2025-02-19T17:24:26.505707Z","shell.execute_reply.started":"2025-02-19T17:24:26.480844Z","shell.execute_reply":"2025-02-19T17:24:26.504858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Explore data img\nimport os\n\n# Count images in train folder\ntrain_images = os.listdir(f\"{data_path}/train\")\ntest_images = os.listdir(f\"{data_path}/test\")\n\nprint(f\"Number of training images: {len(train_images)}\")\nprint(f\"Number of test images: {len(test_images)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:28.283076Z","iopub.execute_input":"2025-02-19T17:24:28.283356Z","iopub.status.idle":"2025-02-19T17:24:28.326914Z","shell.execute_reply.started":"2025-02-19T17:24:28.283335Z","shell.execute_reply":"2025-02-19T17:24:28.326061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fill missing values with the mean of each column\ntrain_df['time_of_event'].fillna(train_df['time_of_event'].mean(), inplace=True)\ntrain_df['time_of_alert'].fillna(train_df['time_of_alert'].mean(), inplace=True)\n\n# Check if there are still missing values\nprint(train_df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:30.200802Z","iopub.execute_input":"2025-02-19T17:24:30.201112Z","iopub.status.idle":"2025-02-19T17:24:30.210395Z","shell.execute_reply.started":"2025-02-19T17:24:30.201087Z","shell.execute_reply":"2025-02-19T17:24:30.209308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the file extensions of images\nimage_extensions = [os.path.splitext(img)[1] for img in train_images]\nprint(set(image_extensions))  # Display unique file extensions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:32.024066Z","iopub.execute_input":"2025-02-19T17:24:32.024412Z","iopub.status.idle":"2025-02-19T17:24:32.031519Z","shell.execute_reply.started":"2025-02-19T17:24:32.024380Z","shell.execute_reply":"2025-02-19T17:24:32.030436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing import image\nimport numpy as np\nimport os\n\n# Image size (e.g., 224x224 for CNN)\nimg_size = (224, 224)\n\n# Load images (only first 10 for now)\nimages = []\nfor img_name in train_images[:10]:  # Load the first 10 images\n    img_path = os.path.join(data_path, 'train', img_name)\n    try:\n        img = image.load_img(img_path, target_size=img_size)  # Resize image\n        img_array = image.img_to_array(img) / 255.0  # Normalize pixel values to [0, 1]\n        images.append(img_array)\n    except Exception as e:\n        print(f\"Error loading image {img_name}: {e}\")\n\n# Convert list to numpy array\nimages = np.array(images)\nprint(images.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:33.679232Z","iopub.execute_input":"2025-02-19T17:24:33.679536Z","iopub.status.idle":"2025-02-19T17:24:48.830577Z","shell.execute_reply.started":"2025-02-19T17:24:33.679511Z","shell.execute_reply":"2025-02-19T17:24:48.829763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install opencv-python\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:48.831733Z","iopub.execute_input":"2025-02-19T17:24:48.832302Z","iopub.status.idle":"2025-02-19T17:24:53.371277Z","shell.execute_reply.started":"2025-02-19T17:24:48.832275Z","shell.execute_reply":"2025-02-19T17:24:53.370209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport os\n\n# Function to extract frames from a video file\ndef extract_frames(video_path, max_frames=10):\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    \n    # Get video properties\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    \n    # If the video has more than max_frames, we pick frames evenly spaced\n    step = total_frames // max_frames\n    \n    for i in range(0, total_frames, step):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, i)\n        ret, frame = cap.read()\n        \n        if ret:\n            # Resize and normalize\n            frame = cv2.resize(frame, (224, 224))  # Resize to 224x224\n            frame = frame / 255.0  # Normalize pixel values to [0, 1]\n            frames.append(frame)\n    \n    cap.release()\n    return frames\n\n# Example: Extract frames from the first video\nvideo_path = os.path.join(data_path, 'train', '02059.mp4')  \nframes = extract_frames(video_path)\nprint(f\"Extracted {len(frames)} frames from {video_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:53.373644Z","iopub.execute_input":"2025-02-19T17:24:53.374012Z","iopub.status.idle":"2025-02-19T17:24:58.195364Z","shell.execute_reply.started":"2025-02-19T17:24:53.373984Z","shell.execute_reply":"2025-02-19T17:24:58.194418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_frames = []\nvideo_names = [video for video in train_images if video.endswith('.mp4')]\n\nfor video_name in video_names[:10]:  \n    video_path = os.path.join(data_path, 'train', video_name)\n    frames = extract_frames(video_path)\n    all_frames.extend(frames)\n\n# Convert list of frames to numpy array\nall_frames = np.array(all_frames)\nprint(f\"Extracted {all_frames.shape[0]} frames in total.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-19T17:24:58.196577Z","iopub.execute_input":"2025-02-19T17:24:58.196871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Analysis (EDA)","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot the distribution of the target variable\ntrain_df['target'].value_counts().plot(kind='bar', color=['skyblue', 'salmon'])\nplt.title(\"Distribution of Target Variable\")\nplt.xlabel(\"Target (0 = No Collision, 1 = Collision)\")\nplt.ylabel(\"Count\")\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation matrix\ncorrelation_matrix = train_df[['time_of_event', 'time_of_alert', 'target']].corr()\n\n# Plotting the correlation matrix\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt='.2f')\nplt.title(\"Correlation Matrix\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_files = [f for f in os.listdir(os.path.join(data_path, 'train')) if f.endswith(('.jpg', '.png'))]\nvideo_files = [f for f in os.listdir(os.path.join(data_path, 'train')) if f.endswith('.mp4')]\n\nprint(f\"Number of image files: {len(image_files)}\")\nprint(f\"Number of video files: {len(video_files)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tensorflow.keras.preprocessing import image\nimport matplotlib.pyplot as plt\n\n# Path to your dataset\ndata_path = '/kaggle/input/nexar-collision-prediction'  # Replace with your actual path\n\n# Get the list of files (skip .mp4 files)\nimage_files = [f for f in os.listdir(os.path.join(data_path, 'train')) if f.endswith(('.jpg', '.png'))]\n\n# Display first 5 images\nfig, axes = plt.subplots(1, 5, figsize=(15, 10))\nfor i, ax in enumerate(axes):\n    img_path = os.path.join(data_path, 'train', image_files[i])  # Use only image files\n    img = image.load_img(img_path, target_size=(224, 224))  # Resize images to 224x224\n    ax.imshow(img)\n    ax.axis('off')\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualizing Sample Extracted Frames","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Display the first 5 extracted frames\nfig, axes = plt.subplots(1, 5, figsize=(15, 10))\nfor i, ax in enumerate(axes):\n    ax.imshow(all_frames[i])\n    ax.axis('off')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}