{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Enhanced Collision Prediction System\nCombines temporal modeling, hybrid features, and ensemble learning","metadata":{"_uuid":"42e09ec1-ff41-4d1c-aa4b-c2bd9fb65579","_cell_guid":"c067c4ec-a463-4dbb-a5ae-eb8f29dd50af","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nimport os\nimport pickle\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score, classification_report, confusion_matrix\nimport joblib\nfrom joblib import Parallel, delayed\nfrom scipy.stats import uniform, randint\nimport warnings\nwarnings.filterwarnings('ignore')\nimport keras_tuner as kt\n\n# Deep Learning\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Bidirectional\nfrom tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, LearningRateScheduler\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\nfrom tensorflow.keras.applications import EfficientNetB0\n\n# Machine Learning\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.model_selection import RandomizedSearchCV\n\nprint(\"Libraries imported successfully!\")\nprint(f\"TensorFlow version: {tf.__version__}\")\nprint(f\"GPU available: {tf.config.list_physical_devices('GPU')}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:07:56.916525Z","iopub.execute_input":"2025-09-30T05:07:56.916729Z","iopub.status.idle":"2025-09-30T05:07:56.98738Z","shell.execute_reply.started":"2025-09-30T05:07:56.916709Z","shell.execute_reply":"2025-09-30T05:07:56.986666Z"}},"outputs":[{"name":"stdout","text":"Libraries imported successfully!\nTensorFlow version: 2.17.1\nGPU available: [PhysicalDevice(name='/physical_device:GPU:0', device_type='GPU')]\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"# Cell 2: Data Loading and Initial Preprocessing\ndef load_and_preprocess_data():\n    \"\"\"Load and preprocess the training data\"\"\"\n    # Load training data\n    train_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\n    \n    # Format IDs to 5-digit strings with leading zeros\n    train_df['id'] = train_df['id'].apply(lambda x: f\"{int(float(x)):05d}\")\n    \n    # Fill missing alert/event times with 0\n    train_df.fillna({'time_of_alert': 0, 'time_of_event': 0}, inplace=True)\n    \n    print(f\"Training data shape: {train_df.shape}\")\n    print(f\"Target distribution:\\n{train_df['target'].value_counts()}\")\n    print(f\"Missing values:\\n{train_df.isnull().sum()}\")\n    \n    return train_df\n\n# Load data\ntrain_df = load_and_preprocess_data()\n\n# Cell 3: Video Frame Extraction Functions\ndef extract_critical_frames(video_path, alert_time, event_time, num_frames=8, sampling_interval=30):\n    \"\"\"\n    Optimized frame extraction without redundant capture\n    \n    Args:\n        video_path: Path to video file\n        alert_time: Time of alert (not used in current implementation)\n        event_time: Time of event (not used in current implementation)  \n        num_frames: Number of frames to extract\n        sampling_interval: Skip frames (sample every nth frame)\n    \n    Returns:\n        numpy array of extracted frames\n    \"\"\"\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    frame_count = 0\n    \n    if not cap.isOpened():\n        print(f\"Warning: Could not open video {video_path}\")\n        return np.zeros((num_frames, 224, 224, 3), dtype=np.uint8)\n    \n    while True:\n        ret, frame = cap.read()\n        if not ret:\n            break\n            \n        # Sample frames at specified interval\n        if frame_count % sampling_interval == 0:\n            # Resize and convert BGR to RGB\n            frame = cv2.resize(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB), (224, 224))\n            frames.append(frame)\n            \n            # Stop when we have enough frames\n            if len(frames) >= num_frames:\n                break\n        frame_count += 1\n    \n    cap.release()\n    \n    # Ensure we return exactly num_frames\n    if len(frames) < num_frames:\n        # Pad with last frame if needed\n        while len(frames) < num_frames:\n            frames.append(frames[-1] if frames else np.zeros((224, 224, 3), dtype=np.uint8))\n    \n    return np.array(frames[:num_frames])\n\ndef calculate_optical_flow(frames):\n    \"\"\"Calculate dense optical flow between consecutive frames\"\"\"\n    flows = []\n    prev_gray = cv2.cvtColor(frames[0], cv2.COLOR_RGB2GRAY)\n    \n    for frame in frames[1:]:\n        gray = cv2.cvtColor(frame, cv2.COLOR_RGB2GRAY)\n        flow = cv2.calcOpticalFlowFarneback(prev_gray, gray, None, 0.5, 3, 15, 3, 5, 1.2, 0)\n        flows.append(np.linalg.norm(flow, axis=2))\n        prev_gray = gray\n    \n    return np.array(flows)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:07:56.988275Z","iopub.execute_input":"2025-09-30T05:07:56.988542Z","iopub.status.idle":"2025-09-30T05:07:57.020522Z","shell.execute_reply.started":"2025-09-30T05:07:56.98851Z","shell.execute_reply":"2025-09-30T05:07:57.019889Z"}},"outputs":[{"name":"stdout","text":"Training data shape: (1500, 4)\nTarget distribution:\ntarget\n0    750\n1    750\nName: count, dtype: int64\nMissing values:\nid               0\ntime_of_event    0\ntime_of_alert    0\ntarget           0\ndtype: int64\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"# Cell 4: Feature Extraction Setup\n# Initialize CNN feature extractor (InceptionV3)\nbase_model = InceptionV3(weights='imagenet', include_top=False, pooling='avg')\ncnn_feature_dim = base_model.output_shape[-1]\n\nprint(f\"CNN feature extractor loaded. Output dimension: {cnn_feature_dim}\")\n\ndef get_hybrid_features(video_path, alert_time, event_time):\n    \"\"\"\n    Extract hybrid features combining spatial (CNN) and temporal (optical flow) features\n    \n    Args:\n        video_path: Path to video file\n        alert_time: Time of alert\n        event_time: Time of event\n    \n    Returns:\n        Combined feature vector (CNN features + flow feature)\n    \"\"\"\n    # Extract frames using optimized method\n    frames = extract_critical_frames(\n        video_path, \n        alert_time, \n        event_time,\n        num_frames=8,\n        sampling_interval=30  # Process 1 frame per second for 30fps video\n    )\n    \n    # Handle case where no frames were extracted\n    if len(frames) == 0 or frames.size == 0:\n        return np.zeros(cnn_feature_dim + 1)  # CNN features + 1 flow feature\n    \n    # Batch process spatial features using CNN\n    try:\n        spatial_features = base_model.predict(\n            preprocess_input(frames.astype('float32')),\n            batch_size=32,\n            verbose=0\n        )\n    except Exception as e:\n        print(f\"Error processing frames for {video_path}: {e}\")\n        return np.zeros(cnn_feature_dim + 1)\n    \n    # Calculate simplified temporal feature (optical flow)\n    flow_feature = 0.0\n    if len(frames) > 1:\n        try:\n            prev_gray = cv2.cvtColor(frames[0], cv2.COLOR_RGB2GRAY)\n            next_gray = cv2.cvtColor(frames[-1], cv2.COLOR_RGB2GRAY)\n            flow = cv2.calcOpticalFlowFarneback(prev_gray, next_gray, None, 0.5, 3, 15, 3, 5, 1.2, 0)\n            flow_feature = np.mean(np.linalg.norm(flow, axis=2))\n        except Exception as e:\n            print(f\"Error calculating optical flow for {video_path}: {e}\")\n            flow_feature = 0.0\n    \n    # Combine spatial (mean of all frames) and temporal features\n    return np.concatenate([\n        np.mean(spatial_features, axis=0),\n        [flow_feature]\n    ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:07:57.02168Z","iopub.execute_input":"2025-09-30T05:07:57.021925Z","iopub.status.idle":"2025-09-30T05:07:58.804212Z","shell.execute_reply.started":"2025-09-30T05:07:57.021905Z","shell.execute_reply":"2025-09-30T05:07:58.803462Z"}},"outputs":[{"name":"stdout","text":"Downloading data from https://storage.googleapis.com/tensorflow/keras-applications/inception_v3/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5\n\u001b[1m87910968/87910968\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 0us/step\nCNN feature extractor loaded. Output dimension: 2048\n","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"# Cell 5: Feature Extraction from Training Videos\ndef extract_training_features(train_df):\n    \"\"\"Extract features from all training videos\"\"\"\n    print(\"Extracting hybrid features from training videos...\")\n    features = []\n    \n    for idx, row in tqdm(train_df.iterrows(), total=len(train_df), desc=\"Processing training videos\"):\n        video_path = f\"/kaggle/input/nexar-collision-prediction/train/{row['id']}.mp4\"\n        \n        if os.path.exists(video_path):\n            feature_vector = get_hybrid_features(\n                video_path, row['time_of_alert'], row['time_of_event']\n            )\n        else:\n            print(f\"Warning: Video {video_path} not found\")\n            feature_vector = np.zeros(cnn_feature_dim + 1)\n        \n        features.append(feature_vector)\n    \n    return np.array(features)\n\n# Extract features\nX = extract_training_features(train_df)\ny = train_df['target'].values\n\nprint(f\"Feature extraction completed!\")\nprint(f\"Feature shape: {X.shape}\")\nprint(f\"Target shape: {y.shape}\")\nprint(f\"Feature statistics:\")\nprint(f\"  Mean: {X.mean():.4f}\")\nprint(f\"  Std: {X.std():.4f}\")\nprint(f\"  Min: {X.min():.4f}\")\nprint(f\"  Max: {X.max():.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:07:58.805204Z","iopub.execute_input":"2025-09-30T05:07:58.805483Z","iopub.status.idle":"2025-09-30T05:23:04.876862Z","shell.execute_reply.started":"2025-09-30T05:07:58.80546Z","shell.execute_reply":"2025-09-30T05:23:04.876107Z"}},"outputs":[{"name":"stdout","text":"Extracting hybrid features from training videos...\n","output_type":"stream"},{"name":"stderr","text":"Processing training videos: 100%|██████████| 1500/1500 [15:06<00:00,  1.66it/s]","output_type":"stream"},{"name":"stdout","text":"Feature extraction completed!\nFeature shape: (1500, 2049)\nTarget shape: (1500,)\nFeature statistics:\n  Mean: 0.5213\n  Std: 0.4572\n  Min: 0.0000\n  Max: 18.7774\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":4},{"cell_type":"code","source":"# Cell 6: Data Splitting and Preprocessing\n# Train-test split with stratification\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, stratify=y, random_state=42\n)\n\nprint(f\"Data split completed:\")\nprint(f\"  Training set: {X_train.shape[0]} samples\")\nprint(f\"  Validation set: {X_val.shape[0]} samples\")\nprint(f\"  Training target distribution: {np.bincount(y_train)}\")\nprint(f\"  Validation target distribution: {np.bincount(y_val)}\")\n\n# Feature scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\n\n# Reshape for LSTM [samples, timesteps, features]\n# Using timesteps=1 since we have one feature vector per video\nX_train_3d = X_train_scaled.reshape((X_train_scaled.shape[0], 1, X_train_scaled.shape[1]))\nX_val_3d = X_val_scaled.reshape((X_val_scaled.shape[0], 1, X_val_scaled.shape[1]))\n\nprint(f\"Data reshaped for LSTM:\")\nprint(f\"  Training 3D shape: {X_train_3d.shape}\")\nprint(f\"  Validation 3D shape: {X_val_3d.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:23:04.877931Z","iopub.execute_input":"2025-09-30T05:23:04.878247Z","iopub.status.idle":"2025-09-30T05:23:04.964323Z","shell.execute_reply.started":"2025-09-30T05:23:04.878216Z","shell.execute_reply":"2025-09-30T05:23:04.963452Z"}},"outputs":[{"name":"stdout","text":"Data split completed:\n  Training set: 1200 samples\n  Validation set: 300 samples\n  Training target distribution: [600 600]\n  Validation target distribution: [150 150]\nData reshaped for LSTM:\n  Training 3D shape: (1200, 1, 2049)\n  Validation 3D shape: (300, 1, 2049)\n","output_type":"stream"}],"execution_count":5},{"cell_type":"code","source":"# Cell 7: Temporal Model Architecture Definition\ndef build_temporal_model(hp):\n    \"\"\"\n    Build temporal model with hyperparameter tuning support\n    \n    Args:\n        hp: Hyperparameter object for tuning\n    \n    Returns:\n        Compiled Keras model\n    \"\"\"\n    model = Sequential([\n        # Bidirectional LSTM layer\n        Bidirectional(LSTM(\n            units=hp.Int(\"lstm_units\", min_value=32, max_value=256, step=32),\n            input_shape=(1, X_train_scaled.shape[1])\n        )),\n        \n        # Dense layer\n        Dense(\n            units=hp.Int(\"dense_units\", min_value=16, max_value=128, step=16),\n            activation=\"relu\"\n        ),\n        \n        # Dropout for regularization\n        Dropout(hp.Float(\"dropout\", min_value=0.2, max_value=0.5, step=0.1)),\n        \n        # Output layer for binary classification\n        Dense(1, activation=\"sigmoid\")\n    ])\n    \n    # Compile model\n    model.compile(\n        loss=\"binary_crossentropy\",\n        optimizer=tf.keras.optimizers.Adam(\n            hp.Choice(\"learning_rate\", values=[1e-2, 1e-3, 1e-4])\n        ),\n        metrics=[\"accuracy\", tf.keras.metrics.AUC()]\n    )\n    \n    return model\n\n# Enable mixed precision for faster training\ntf.keras.mixed_precision.set_global_policy('mixed_float16')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:23:04.965209Z","iopub.execute_input":"2025-09-30T05:23:04.965546Z","iopub.status.idle":"2025-09-30T05:23:04.971107Z","shell.execute_reply.started":"2025-09-30T05:23:04.965515Z","shell.execute_reply":"2025-09-30T05:23:04.97036Z"}},"outputs":[],"execution_count":6},{"cell_type":"code","source":"# Cell 8: Hyperparameter Tuning for Temporal Model\n# Initialize the tuner\ntuner = kt.Hyperband(\n    build_temporal_model,\n    objective=\"val_auc\",  # Changed to AUC as it's more suitable for imbalanced data\n    max_epochs=20,\n    factor=3,\n    directory=\"kt_logs\",\n    project_name=\"temporal_model_tuning\"\n)\n\nprint(\"Starting hyperparameter tuning...\")\n\n# Perform hyperparameter search\ntuner.search(\n    X_train_3d, y_train,\n    validation_data=(X_val_3d, y_val),\n    epochs=20,\n    batch_size=64,\n    callbacks=[tf.keras.callbacks.EarlyStopping(patience=5, monitor='val_auc', mode='max')]\n)\n\n# Get best hyperparameters and model\nbest_hps = tuner.get_best_hyperparameters(num_trials=1)[0]\nbest_temporal_model = tuner.get_best_models(num_models=1)[0]\n\nprint(\"Best hyperparameters found:\")\nprint(f\"  LSTM units: {best_hps.get('lstm_units')}\")\nprint(f\"  Dense units: {best_hps.get('dense_units')}\")\nprint(f\"  Dropout: {best_hps.get('dropout')}\")\nprint(f\"  Learning rate: {best_hps.get('learning_rate')}\")\n\nbest_temporal_model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:23:04.972926Z","iopub.execute_input":"2025-09-30T05:23:04.973149Z","iopub.status.idle":"2025-09-30T05:23:05.608969Z","shell.execute_reply.started":"2025-09-30T05:23:04.973129Z","shell.execute_reply":"2025-09-30T05:23:05.608271Z"}},"outputs":[{"name":"stdout","text":"Reloading Tuner from kt_logs/temporal_model_tuning/tuner0.json\nStarting hyperparameter tuning...\nBest hyperparameters found:\n  LSTM units: 192\n  Dense units: 80\n  Dropout: 0.2\n  Learning rate: 0.001\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"\u001b[1mModel: \"sequential\"\u001b[0m\n","text/html":"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"font-weight: bold\">Model: \"sequential\"</span>\n</pre>\n"},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓\n┃\u001b[1m \u001b[0m\u001b[1mLayer (type)                        \u001b[0m\u001b[1m \u001b[0m┃\u001b[1m \u001b[0m\u001b[1mOutput Shape               \u001b[0m\u001b[1m \u001b[0m┃\u001b[1m \u001b[0m\u001b[1m        Param #\u001b[0m\u001b[1m \u001b[0m┃\n┡━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩\n│ bidirectional (\u001b[38;5;33mBidirectional\u001b[0m)        │ (\u001b[38;5;45mNone\u001b[0m, \u001b[38;5;34m384\u001b[0m)                 │       \u001b[38;5;34m3,443,712\u001b[0m │\n├──────────────────────────────────────┼─────────────────────────────┼─────────────────┤\n│ dense (\u001b[38;5;33mDense\u001b[0m)                        │ (\u001b[38;5;45mNone\u001b[0m, \u001b[38;5;34m80\u001b[0m)                  │          \u001b[38;5;34m30,800\u001b[0m │\n├──────────────────────────────────────┼─────────────────────────────┼─────────────────┤\n│ dropout (\u001b[38;5;33mDropout\u001b[0m)                    │ (\u001b[38;5;45mNone\u001b[0m, \u001b[38;5;34m80\u001b[0m)                  │               \u001b[38;5;34m0\u001b[0m │\n├──────────────────────────────────────┼─────────────────────────────┼─────────────────┤\n│ dense_1 (\u001b[38;5;33mDense\u001b[0m)                      │ (\u001b[38;5;45mNone\u001b[0m, \u001b[38;5;34m1\u001b[0m)                   │              \u001b[38;5;34m81\u001b[0m │\n└──────────────────────────────────────┴─────────────────────────────┴─────────────────┘\n","text/html":"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\">┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓\n┃<span style=\"font-weight: bold\"> Layer (type)                         </span>┃<span style=\"font-weight: bold\"> Output Shape                </span>┃<span style=\"font-weight: bold\">         Param # </span>┃\n┡━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩\n│ bidirectional (<span style=\"color: #0087ff; text-decoration-color: #0087ff\">Bidirectional</span>)        │ (<span style=\"color: #00d7ff; text-decoration-color: #00d7ff\">None</span>, <span style=\"color: #00af00; text-decoration-color: #00af00\">384</span>)                 │       <span style=\"color: #00af00; text-decoration-color: #00af00\">3,443,712</span> │\n├──────────────────────────────────────┼─────────────────────────────┼─────────────────┤\n│ dense (<span style=\"color: #0087ff; text-decoration-color: #0087ff\">Dense</span>)                        │ (<span style=\"color: #00d7ff; text-decoration-color: #00d7ff\">None</span>, <span style=\"color: #00af00; text-decoration-color: #00af00\">80</span>)                  │          <span style=\"color: #00af00; text-decoration-color: #00af00\">30,800</span> │\n├──────────────────────────────────────┼─────────────────────────────┼─────────────────┤\n│ dropout (<span style=\"color: #0087ff; text-decoration-color: #0087ff\">Dropout</span>)                    │ (<span style=\"color: #00d7ff; text-decoration-color: #00d7ff\">None</span>, <span style=\"color: #00af00; text-decoration-color: #00af00\">80</span>)                  │               <span style=\"color: #00af00; text-decoration-color: #00af00\">0</span> │\n├──────────────────────────────────────┼─────────────────────────────┼─────────────────┤\n│ dense_1 (<span style=\"color: #0087ff; text-decoration-color: #0087ff\">Dense</span>)                      │ (<span style=\"color: #00d7ff; text-decoration-color: #00d7ff\">None</span>, <span style=\"color: #00af00; text-decoration-color: #00af00\">1</span>)                   │              <span style=\"color: #00af00; text-decoration-color: #00af00\">81</span> │\n└──────────────────────────────────────┴─────────────────────────────┴─────────────────┘\n</pre>\n"},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"\u001b[1m Total params: \u001b[0m\u001b[38;5;34m3,474,593\u001b[0m (13.25 MB)\n","text/html":"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"font-weight: bold\"> Total params: </span><span style=\"color: #00af00; text-decoration-color: #00af00\">3,474,593</span> (13.25 MB)\n</pre>\n"},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"\u001b[1m Trainable params: \u001b[0m\u001b[38;5;34m3,474,593\u001b[0m (13.25 MB)\n","text/html":"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"font-weight: bold\"> Trainable params: </span><span style=\"color: #00af00; text-decoration-color: #00af00\">3,474,593</span> (13.25 MB)\n</pre>\n"},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"\u001b[1m Non-trainable params: \u001b[0m\u001b[38;5;34m0\u001b[0m (0.00 B)\n","text/html":"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"font-weight: bold\"> Non-trainable params: </span><span style=\"color: #00af00; text-decoration-color: #00af00\">0</span> (0.00 B)\n</pre>\n"},"metadata":{}}],"execution_count":7},{"cell_type":"code","source":"# Cell 9: Train the Best Temporal Model\nprint(\"Training the best temporal model...\")\n\n# Train the model\nhistory = best_temporal_model.fit(\n    X_train_3d, y_train,\n    validation_data=(X_val_3d, y_val),\n    epochs=30,\n    batch_size=64,\n    callbacks=[\n        EarlyStopping(patience=5, restore_best_weights=True, monitor='val_auc', mode='max'),\n        LearningRateScheduler(lambda epoch: best_hps.get('learning_rate') * (0.95 ** epoch))\n    ],\n    verbose=1\n)\n\n# Evaluate temporal model\ntrain_probs = best_temporal_model.predict(X_train_3d, verbose=0).flatten()\nval_probs = best_temporal_model.predict(X_val_3d, verbose=0).flatten()\n\ntemporal_train_auc = roc_auc_score(y_train, train_probs)\ntemporal_val_auc = roc_auc_score(y_val, val_probs)\n\nprint(f\"\\nTemporal Model Performance:\")\nprint(f\"  Training AUC: {temporal_train_auc:.4f}\")\nprint(f\"  Validation AUC: {temporal_val_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:23:05.610109Z","iopub.execute_input":"2025-09-30T05:23:05.610319Z","iopub.status.idle":"2025-09-30T05:23:13.288245Z","shell.execute_reply.started":"2025-09-30T05:23:05.610301Z","shell.execute_reply":"2025-09-30T05:23:13.287473Z"}},"outputs":[{"name":"stdout","text":"Training the best temporal model...\nEpoch 1/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m6s\u001b[0m 29ms/step - accuracy: 0.9450 - auc: 0.9840 - loss: 0.2078 - val_accuracy: 0.5867 - val_auc: 0.6591 - val_loss: 0.8607 - learning_rate: 0.0010\nEpoch 2/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 8ms/step - accuracy: 0.9722 - auc: 0.9967 - loss: 0.1054 - val_accuracy: 0.6033 - val_auc: 0.6588 - val_loss: 0.9420 - learning_rate: 9.5000e-04\nEpoch 3/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 9ms/step - accuracy: 0.9908 - auc: 0.9998 - loss: 0.0502 - val_accuracy: 0.6133 - val_auc: 0.6673 - val_loss: 1.0750 - learning_rate: 9.0250e-04\nEpoch 4/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 8ms/step - accuracy: 0.9967 - auc: 1.0000 - loss: 0.0183 - val_accuracy: 0.5933 - val_auc: 0.6433 - val_loss: 1.1423 - learning_rate: 8.5737e-04\nEpoch 5/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 8ms/step - accuracy: 1.0000 - auc: 1.0000 - loss: 0.0099 - val_accuracy: 0.6233 - val_auc: 0.6662 - val_loss: 1.2453 - learning_rate: 8.1451e-04\nEpoch 6/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 8ms/step - accuracy: 1.0000 - auc: 1.0000 - loss: 0.0043 - val_accuracy: 0.6067 - val_auc: 0.6476 - val_loss: 1.3254 - learning_rate: 7.7378e-04\nEpoch 7/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 8ms/step - accuracy: 1.0000 - auc: 1.0000 - loss: 0.0024 - val_accuracy: 0.6067 - val_auc: 0.6528 - val_loss: 1.3630 - learning_rate: 7.3509e-04\nEpoch 8/30\n\u001b[1m19/19\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 8ms/step - accuracy: 1.0000 - auc: 1.0000 - loss: 0.0021 - val_accuracy: 0.6000 - val_auc: 0.6491 - val_loss: 1.3919 - learning_rate: 6.9834e-04\n\nTemporal Model Performance:\n  Training AUC: 1.0000\n  Validation AUC: 0.6665\n","output_type":"stream"}],"execution_count":8},{"cell_type":"code","source":"# Create ensemble features (temporal predictions + original features)\nX_train_ensemble = np.column_stack([train_probs, X_train_scaled])\nX_val_ensemble = np.column_stack([val_probs, X_val_scaled])\n\nprint(f\"Ensemble feature shape: {X_train_ensemble.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:23:13.289002Z","iopub.execute_input":"2025-09-30T05:23:13.28921Z","iopub.status.idle":"2025-09-30T05:23:13.297061Z","shell.execute_reply.started":"2025-09-30T05:23:13.289192Z","shell.execute_reply":"2025-09-30T05:23:13.296199Z"}},"outputs":[{"name":"stdout","text":"Ensemble feature shape: (1200, 2050)\n","output_type":"stream"}],"execution_count":9},{"cell_type":"code","source":"# Cell 11: XGBoost Hyperparameter Optimization\n# Define hyperparameter search space for XGBoost\nparam_dist = {\n    'learning_rate': uniform(0.01, 0.29),  # 0.01 to 0.3\n    'max_depth': randint(3, 10),\n    'subsample': uniform(0.6, 0.4),  # 0.6 to 1.0\n    'colsample_bytree': uniform(0.6, 0.4),  # 0.6 to 1.0\n    'gamma': uniform(0, 0.5),\n    'reg_alpha': uniform(0, 1),\n    'reg_lambda': uniform(0, 1),\n    'n_estimators': randint(100, 500)\n}\n\n# Create randomized search\nprint(\"Starting XGBoost hyperparameter optimization...\")\n\nxgb_optimizer = RandomizedSearchCV(\n    estimator=XGBClassifier(\n        objective='binary:logistic',\n        eval_metric='auc',\n        use_label_encoder=False,\n        random_state=42\n    ),\n    param_distributions=param_dist,\n    n_iter=50,  # Number of parameter combinations to try\n    scoring='roc_auc',\n    cv=3,  # 3-fold cross-validation\n    n_jobs=-1,\n    verbose=1,\n    random_state=42\n)\n\n# Run optimization\nxgb_optimizer.fit(X_train_ensemble, y_train)\n\nprint(\"XGBoost optimization completed!\")\nprint(\"Best parameters:\", xgb_optimizer.best_params_)\nprint(f\"Best CV score: {xgb_optimizer.best_score_:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:23:13.297972Z","iopub.execute_input":"2025-09-30T05:23:13.298279Z","iopub.status.idle":"2025-09-30T05:28:44.390709Z","shell.execute_reply.started":"2025-09-30T05:23:13.298248Z","shell.execute_reply":"2025-09-30T05:28:44.389021Z"}},"outputs":[{"name":"stdout","text":"Starting XGBoost hyperparameter optimization...\nFitting 3 folds for each of 50 candidates, totalling 150 fits\nXGBoost optimization completed!\nBest parameters: {'colsample_bytree': 0.9659838702175123, 'gamma': 0.42501928889489965, 'learning_rate': 0.140340695500079, 'max_depth': 3, 'n_estimators': 147, 'reg_alpha': 0.37081825219826636, 'reg_lambda': 0.6688412526636073, 'subsample': 0.8663689426469987}\nBest CV score: 1.0000\n","output_type":"stream"}],"execution_count":10},{"cell_type":"code","source":"# Cell 12: Train Final Ensemble Model\n# Get the best XGBoost model\nbest_xgb = xgb_optimizer.best_estimator_\n\n# Train with early stopping\nprint(\"Training final ensemble model...\")\n\nbest_xgb.fit(\n    X_train_ensemble, y_train,\n    eval_set=[(X_val_ensemble, y_val)],\n    early_stopping_rounds=20,\n    verbose=False\n)\n\n# Generate ensemble predictions\nensemble_train_probs = best_xgb.predict_proba(X_train_ensemble)[:, 1]\nensemble_val_probs = best_xgb.predict_proba(X_val_ensemble)[:, 1]\n\n# Calculate final metrics\nensemble_train_auc = roc_auc_score(y_train, ensemble_train_probs)\nensemble_val_auc = roc_auc_score(y_val, ensemble_val_probs)\n\nprint(f\"\\nFinal Ensemble Model Performance:\")\nprint(f\"  Training AUC: {ensemble_train_auc:.4f}\")\nprint(f\"  Validation AUC: {ensemble_val_auc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:28:44.392091Z","iopub.execute_input":"2025-09-30T05:28:44.393182Z","iopub.status.idle":"2025-09-30T05:28:45.351499Z","shell.execute_reply.started":"2025-09-30T05:28:44.393148Z","shell.execute_reply":"2025-09-30T05:28:45.350758Z"}},"outputs":[{"name":"stdout","text":"Training final ensemble model...\n\nFinal Ensemble Model Performance:\n  Training AUC: 1.0000\n  Validation AUC: 0.6333\n","output_type":"stream"}],"execution_count":11},{"cell_type":"code","source":"# Cell 13: Model Evaluation and Classification Report\n# Convert probabilities to binary predictions using 0.5 threshold\nensemble_val_pred = (ensemble_val_probs >= 0.5).astype(int)\n\nprint(\"\\nDetailed Classification Report:\")\nprint(classification_report(y_val, ensemble_val_pred))\n\nprint(\"\\nConfusion Matrix:\")\ncm = confusion_matrix(y_val, ensemble_val_pred)\nprint(cm)\n\n# Calculate additional metrics\nfrom sklearn.metrics import precision_score, recall_score, f1_score\n\nprecision = precision_score(y_val, ensemble_val_pred)\nrecall = recall_score(y_val, ensemble_val_pred)\nf1 = f1_score(y_val, ensemble_val_pred)\n\nprint(f\"\\nAdditional Metrics:\")\nprint(f\"  Precision: {precision:.4f}\")\nprint(f\"  Recall: {recall:.4f}\")\nprint(f\"  F1-score: {f1:.4f}\")\nprint(f\"  AUC-ROC: {ensemble_val_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:28:45.352289Z","iopub.execute_input":"2025-09-30T05:28:45.352531Z","iopub.status.idle":"2025-09-30T05:28:45.372675Z","shell.execute_reply.started":"2025-09-30T05:28:45.352511Z","shell.execute_reply":"2025-09-30T05:28:45.371876Z"}},"outputs":[{"name":"stdout","text":"\nDetailed Classification Report:\n              precision    recall  f1-score   support\n\n           0       0.62      0.67      0.65       150\n           1       0.64      0.59      0.62       150\n\n    accuracy                           0.63       300\n   macro avg       0.63      0.63      0.63       300\nweighted avg       0.63      0.63      0.63       300\n\n\nConfusion Matrix:\n[[101  49]\n [ 61  89]]\n\nAdditional Metrics:\n  Precision: 0.6449\n  Recall: 0.5933\n  F1-score: 0.6181\n  AUC-ROC: 0.6333\n","output_type":"stream"}],"execution_count":12},{"cell_type":"code","source":"# Cell 14: Save Models and Preprocessing Objects\nprint(\"Saving models and preprocessing objects...\")\n\n# Save the scaler\njoblib.dump(scaler, 'collision_prediction_scaler.pkl')\nprint(\"✓ Scaler saved as 'collision_prediction_scaler.pkl'\")\n\n# Save the temporal model\nbest_temporal_model.save('collision_prediction_temporal_model.h5')\nprint(\"✓ Temporal model saved as 'collision_prediction_temporal_model.h5'\")\n\n# Save the XGBoost ensemble model\njoblib.dump(best_xgb, 'collision_prediction_ensemble_model.pkl')\nprint(\"✓ Ensemble model saved as 'collision_prediction_ensemble_model.pkl'\")\n\n# Save the CNN feature extractor info (we'll need to recreate it during inference)\nmodel_info = {\n    'cnn_feature_dim': cnn_feature_dim,\n    'best_hyperparameters': {\n        'lstm_units': best_hps.get('lstm_units'),\n        'dense_units': best_hps.get('dense_units'),\n        'dropout': best_hps.get('dropout'),\n        'learning_rate': best_hps.get('learning_rate')\n    },\n    'feature_shape': X_train_scaled.shape[1],\n    'ensemble_shape': X_train_ensemble.shape[1]\n}\n\nwith open('model_info.pkl', 'wb') as f:\n    pickle.dump(model_info, f)\nprint(\"✓ Model info saved as 'model_info.pkl'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:28:45.373544Z","iopub.execute_input":"2025-09-30T05:28:45.373825Z","iopub.status.idle":"2025-09-30T05:28:45.508521Z","shell.execute_reply.started":"2025-09-30T05:28:45.373802Z","shell.execute_reply":"2025-09-30T05:28:45.507674Z"}},"outputs":[{"name":"stdout","text":"Saving models and preprocessing objects...\n✓ Scaler saved as 'collision_prediction_scaler.pkl'\n✓ Temporal model saved as 'collision_prediction_temporal_model.h5'\n✓ Ensemble model saved as 'collision_prediction_ensemble_model.pkl'\n✓ Model info saved as 'model_info.pkl'\n","output_type":"stream"}],"execution_count":13},{"cell_type":"code","source":"# Cell 15: Test Set Processing and Prediction\ndef process_test_data():\n    \"\"\"Process test data and generate predictions\"\"\"\n    print(\"Loading test data...\")\n    test_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\n    test_df['id'] = test_df['id'].apply(lambda x: f\"{int(float(x)):05d}\")\n    \n    print(\"Extracting features from test videos...\")\n    test_features = []\n    \n    for idx, row in tqdm(test_df.iterrows(), total=len(test_df), desc=\"Processing test videos\"):\n        video_path = f\"/kaggle/input/nexar-collision-prediction/test/{row['id']}.mp4\"\n        \n        if os.path.exists(video_path):\n            feature_vector = get_hybrid_features(video_path, 0, 0)  # No event times in test\n        else:\n            print(f\"Warning: Test video {video_path} not found\")\n            feature_vector = np.zeros(cnn_feature_dim + 1)\n        \n        test_features.append(feature_vector)\n    \n    X_test = np.array(test_features)\n    \n    # Apply same preprocessing\n    X_test_scaled = scaler.transform(X_test)\n    X_test_3d = X_test_scaled.reshape((X_test_scaled.shape[0], 1, X_test_scaled.shape[1]))\n    \n    # Generate temporal predictions\n    temporal_test_probs = best_temporal_model.predict(X_test_3d, verbose=0).flatten()\n    \n    # Create ensemble features\n    X_test_ensemble = np.column_stack([temporal_test_probs, X_test_scaled])\n    \n    # Generate final predictions\n    final_test_probs = best_xgb.predict_proba(X_test_ensemble)[:, 1]\n    \n    return test_df, final_test_probs\n\n# Process test data\ntest_df, final_predictions = process_test_data()\n\nprint(f\"Test predictions generated for {len(test_df)} videos\")\nprint(f\"Prediction statistics:\")\nprint(f\"  Mean: {final_predictions.mean():.4f}\")\nprint(f\"  Std: {final_predictions.std():.4f}\")\nprint(f\"  Min: {final_predictions.min():.4f}\")\nprint(f\"  Max: {final_predictions.max():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:28:45.509112Z","iopub.execute_input":"2025-09-30T05:28:45.509335Z","iopub.status.idle":"2025-09-30T05:41:38.537871Z","shell.execute_reply.started":"2025-09-30T05:28:45.509314Z","shell.execute_reply":"2025-09-30T05:41:38.536896Z"}},"outputs":[{"name":"stdout","text":"Loading test data...\nExtracting features from test videos...\n","output_type":"stream"},{"name":"stderr","text":"Processing test videos: 100%|██████████| 1344/1344 [12:52<00:00,  1.74it/s]","output_type":"stream"},{"name":"stdout","text":"Test predictions generated for 1344 videos\nPrediction statistics:\n  Mean: 0.4926\n  Std: 0.0689\n  Min: 0.4307\n  Max: 0.5693\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":14},{"cell_type":"code","source":"# Cell 16: Create Submission File\n# Create submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test_df['id'],\n    'score': final_predictions\n})\n\n# Save submission file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully!\")\nprint(\"\\nSubmission Summary:\")\nprint(submission.describe())\nprint(f\"\\nFirst few predictions:\")\nprint(submission.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:41:38.538657Z","iopub.execute_input":"2025-09-30T05:41:38.538897Z","iopub.status.idle":"2025-09-30T05:41:38.565511Z","shell.execute_reply.started":"2025-09-30T05:41:38.538876Z","shell.execute_reply":"2025-09-30T05:41:38.564915Z"}},"outputs":[{"name":"stdout","text":"Submission file created successfully!\n\nSubmission Summary:\n             score\ncount  1344.000000\nmean      0.492587\nstd       0.068887\nmin       0.430747\n25%       0.430747\n50%       0.430747\n75%       0.569269\nmax       0.569269\n\nFirst few predictions:\n      id     score\n0  00204  0.430747\n1  00030  0.430747\n2  00146  0.430747\n3  00020  0.569269\n4  00511  0.569269\n5  00261  0.569269\n6  00346  0.569269\n7  00545  0.569269\n8  00492  0.430747\n9  00299  0.430747\n","output_type":"stream"}],"execution_count":15},{"cell_type":"code","source":"# Cell 17: Complete Model Package for Deployment\ndef create_deployment_package():\n    \"\"\"Create a complete package for model deployment\"\"\"\n    \n    deployment_info = {\n        'model_files': {\n            'scaler': 'collision_prediction_scaler.pkl',\n            'temporal_model': 'collision_prediction_temporal_model.h5', \n            'ensemble_model': 'collision_prediction_ensemble_model.pkl',\n            'model_info': 'model_info.pkl'\n        },\n        'preprocessing_params': {\n            'num_frames': 8,\n            'sampling_interval': 30,\n            'frame_size': (224, 224),\n            'cnn_model': 'InceptionV3'\n        },\n        'performance_metrics': {\n            'temporal_model_auc': temporal_val_auc,\n            'ensemble_model_auc': ensemble_val_auc,\n            'precision': precision,\n            'recall': recall,\n            'f1_score': f1\n        },\n        'feature_info': {\n            'total_features': X_train_scaled.shape[1],\n            'cnn_features': cnn_feature_dim,\n            'temporal_features': 1,\n            'ensemble_features': X_train_ensemble.shape[1]\n        }\n    }\n    \n    with open('deployment_info.pkl', 'wb') as f:\n        pickle.dump(deployment_info, f)\n    \n    print(\"✓ Deployment package created!\")\n    print(\"\\nFiles for deployment:\")\n    for key, filename in deployment_info['model_files'].items():\n        print(f\"  {key}: {filename}\")\n    print(\"  deployment_info.pkl\")\n    \n    return deployment_info\n\n# Create deployment package\ndeployment_package = create_deployment_package()\n\nprint(\"\\n\" + \"=\"*50)\nprint(\"COLLISION PREDICTION MODEL TRAINING COMPLETED!\")\nprint(\"=\"*50)\nprint(f\"Final Model Performance:\")\nprint(f\"  Validation AUC: {ensemble_val_auc:.4f}\")\nprint(f\"  Precision: {precision:.4f}\")\nprint(f\"  Recall: {recall:.4f}\")\nprint(f\"  F1-score: {f1:.4f}\")\nprint(\"\\nAll models and preprocessing objects saved successfully!\")\nprint(\"Ready for deployment in AI agent project.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-30T05:41:38.566214Z","iopub.execute_input":"2025-09-30T05:41:38.566405Z","iopub.status.idle":"2025-09-30T05:41:38.581548Z","shell.execute_reply.started":"2025-09-30T05:41:38.566386Z","shell.execute_reply":"2025-09-30T05:41:38.580717Z"}},"outputs":[{"name":"stdout","text":"✓ Deployment package created!\n\nFiles for deployment:\n  scaler: collision_prediction_scaler.pkl\n  temporal_model: collision_prediction_temporal_model.h5\n  ensemble_model: collision_prediction_ensemble_model.pkl\n  model_info: model_info.pkl\n  deployment_info.pkl\n\n==================================================\nCOLLISION PREDICTION MODEL TRAINING COMPLETED!\n==================================================\nFinal Model Performance:\n  Validation AUC: 0.6333\n  Precision: 0.6449\n  Recall: 0.5933\n  F1-score: 0.6181\n\nAll models and preprocessing objects saved successfully!\nReady for deployment in AI agent project.\n","output_type":"stream"}],"execution_count":16}]}