{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Enhanced Collision Prediction System\nCombines temporal modeling, hybrid features, and ensemble learning","metadata":{"_uuid":"42e09ec1-ff41-4d1c-aa4b-c2bd9fb65579","_cell_guid":"c067c4ec-a463-4dbb-a5ae-eb8f29dd50af","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Core Imports\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_auc_score\nimport joblib\nfrom joblib import Parallel, delayed\nfrom scipy.stats import uniform, randint\nimport warnings\nwarnings.filterwarnings('ignore')\nimport keras_tuner as kt\n\n# Deep Learning\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Bidirectional\nfrom tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, LearningRateScheduler\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\nfrom tensorflow.keras.applications import EfficientNetB0\n\n# Machine Learning\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.model_selection import RandomizedSearchCV","metadata":{"_uuid":"621419f6-981c-4400-ab19-80b8e5c8601e","_cell_guid":"b0e957f7-f9a4-4fbf-902b-2dd40871d9ef","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:16:50.568866Z","iopub.execute_input":"2025-03-05T18:16:50.569178Z","iopub.status.idle":"2025-03-05T18:16:51.038278Z","shell.execute_reply.started":"2025-03-05T18:16:50.569154Z","shell.execute_reply":"2025-03-05T18:16:51.037377Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Enhanced Frame Sampling & Feature Extraction","metadata":{"_uuid":"eafc7ccf-0254-46a6-a02a-4ec812afe0e8","_cell_guid":"d493259c-c194-459d-a5fe-b48d9eeb1cbb","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def extract_critical_frames(video_path, alert_time, event_time, num_frames=8, sampling_interval=30):\n    \"\"\"Optimized frame extraction without redundant capture\"\"\"\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    frame_count = 0\n    \n    while True:\n        ret, frame = cap.read()\n        if not ret:\n            break\n        if frame_count % sampling_interval == 0:\n            frame = cv2.resize(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB), (224, 224))\n            frames.append(frame)\n            if len(frames) >= num_frames:\n                break\n        frame_count += 1\n    \n    cap.release()\n    return np.array(frames[:num_frames])  # Return exactly num_frames\n\ndef calculate_optical_flow(frames):\n    \"\"\"Calculate dense optical flow between consecutive frames\"\"\"\n    flows = []\n    prev_gray = cv2.cvtColor(frames[0], cv2.COLOR_RGB2GRAY)\n    \n    for frame in frames[1:]:\n        gray = cv2.cvtColor(frame, cv2.COLOR_RGB2GRAY)\n        flow = cv2.calcOpticalFlowFarneback(prev_gray, gray, None, 0.5, 3, 15, 3, 5, 1.2, 0)\n        flows.append(np.linalg.norm(flow, axis=2))\n        prev_gray = gray\n    \n    return np.array(flows)","metadata":{"_uuid":"0b8b7193-8825-4103-a95c-d0266248062d","_cell_guid":"662588b9-86d9-42b0-8626-a63fc5604c49","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:15:56.985335Z","iopub.execute_input":"2025-03-05T18:15:56.985643Z","iopub.status.idle":"2025-03-05T18:15:56.992106Z","shell.execute_reply.started":"2025-03-05T18:15:56.985620Z","shell.execute_reply":"2025-03-05T18:15:56.991315Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Hybrid Feature Engineering","metadata":{"_uuid":"875d930b-58c9-4852-99a0-7f0765cf1860","_cell_guid":"500e553d-e707-40c1-ac45-9fcc035753af","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Initialize feature extractor\nbase_model = InceptionV3(weights='imagenet', include_top=False, pooling='avg')\ncnn_feature_dim = base_model.output_shape[-1]\n\ndef get_hybrid_features(video_path, alert_time, event_time):\n    \"\"\"Optimized feature extraction with proper resource handling\"\"\"\n    # Use optimized frame extraction\n    frames = extract_critical_frames(\n        video_path, \n        alert_time, \n        event_time,\n        num_frames=8,  # Reduced from 16\n        sampling_interval=30  # Process 1 frame per second (30fps video)\n    )\n    \n    if len(frames) == 0:\n        return np.zeros(1280 + 1)  # EfficientNetB0 features + flow feature\n    \n    # Batch process spatial features\n    spatial_features = base_model.predict(\n        preprocess_input(frames.astype('float32')),\n        batch_size=32,  # Process 32 frames at once\n        verbose=0\n    )\n    \n    # Simplified temporal feature\n    flow_feature = 0.0\n    if len(frames) > 1:\n        prev_gray = cv2.cvtColor(frames[0], cv2.COLOR_RGB2GRAY)\n        next_gray = cv2.cvtColor(frames[-1], cv2.COLOR_RGB2GRAY)\n        flow = cv2.calcOpticalFlowFarneback(prev_gray, next_gray, None, 0.5, 3, 15, 3, 5, 1.2, 0)\n        flow_feature = np.mean(np.linalg.norm(flow, axis=2))\n    \n    return np.concatenate([\n        np.mean(spatial_features, axis=0),\n        [flow_feature]\n    ])","metadata":{"_uuid":"fcb5067b-2463-4b4e-ae8d-f8d87f61769f","_cell_guid":"bf618b97-52e5-4caa-97ac-c04d747dc9c4","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:16:05.009034Z","iopub.execute_input":"2025-03-05T18:16:05.009382Z","iopub.status.idle":"2025-03-05T18:16:13.355269Z","shell.execute_reply.started":"2025-03-05T18:16:05.009354Z","shell.execute_reply":"2025-03-05T18:16:13.354315Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Temporal Model Architecture","metadata":{"_uuid":"6dafe37e-35c2-422f-9e30-7086597fe89e","_cell_guid":"754e12b7-c6be-41c0-97d7-31bcf38fc50b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def build_temporal_model(hp):\n    model = Sequential([\n        # Tune the number of LSTM units\n        Bidirectional(LSTM(\n            units=hp.Int(\"lstm_units\", min_value=32, max_value=256, step=32),\n            input_shape=(1, X_train_scaled.shape[1])\n        )),\n        \n        # Fully connected layer (Dense)\n        Dense(\n            units=hp.Int(\"dense_units\", min_value=16, max_value=128, step=16),\n            activation=\"relu\"\n        ),\n        \n        # Dropout for regularization\n        Dropout(hp.Float(\"dropout\", min_value=0.2, max_value=0.5, step=0.1)),\n        \n        # Output layer\n        Dense(1, activation=\"sigmoid\")\n    ])\n    \n    # Compile the model\n    model.compile(\n        loss=\"binary_crossentropy\",\n        optimizer=tf.keras.optimizers.Adam(\n            hp.Choice(\"learning_rate\", values=[1e-2, 1e-3, 1e-4])\n        ),\n        metrics=[\"accuracy\", tf.keras.metrics.AUC()]\n    )\n    \n    return model","metadata":{"_uuid":"826de64f-6d31-46e2-b260-47d33b40a1d3","_cell_guid":"549db99d-2057-4437-b4df-d2291661cd65","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:17:32.893846Z","iopub.execute_input":"2025-03-05T18:17:32.894127Z","iopub.status.idle":"2025-03-05T18:17:32.899571Z","shell.execute_reply.started":"2025-03-05T18:17:32.894107Z","shell.execute_reply":"2025-03-05T18:17:32.898603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and preprocess data\ntrain_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\ntrain_df['id'] = train_df['id'].apply(lambda x: f\"{int(float(x)):05d}\")\ntrain_df.fillna({'time_of_alert': 0, 'time_of_event': 0}, inplace=True)\n\n# Feature extraction\nprint(\"Extracting hybrid features...\")\nfeatures = []\nfor _, row in tqdm(train_df.iterrows(), total=len(train_df), desc=\"Hybrid features extracted:\"):\n    video_path = f\"/kaggle/input/nexar-collision-prediction/train/{row['id']}.mp4\"\n    features.append(get_hybrid_features(\n        video_path, row['time_of_alert'], row['time_of_event']\n    ))\n\nX = np.array(features)\ny = train_df['target'].values\n\n# Train-test split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, stratify=y)\n\n# Temporal model training\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\n\n# Reshape for LSTM [samples, timesteps, features]\nX_train_3d = X_train_scaled.reshape((X_train_scaled.shape[0], 1, X_train_scaled.shape[1]))\nX_val_3d = X_val_scaled.reshape((X_val_scaled.shape[0], 1, X_val_scaled.shape[1]))\n\ntf.keras.mixed_precision.set_global_policy('mixed_float16')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T18:18:05.257702Z","iopub.execute_input":"2025-03-05T18:18:05.257989Z","iopub.status.idle":"2025-03-05T18:33:50.410733Z","shell.execute_reply.started":"2025-03-05T18:18:05.257967Z","shell.execute_reply":"2025-03-05T18:33:50.409778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize the tuner\ntuner = kt.Hyperband(\n    build_temporal_model,\n    objective=\"val_accuracy\",\n    max_epochs=20,\n    factor=3,\n    directory=\"kt_logs\",\n    project_name=\"temporal_model_tuning\"\n)\n\n# Perform hyperparameter search\ntuner.search(\n    X_train_3d, y_train,\n    validation_data=(X_val_3d, y_val),\n    epochs=20,\n    batch_size=64,\n    callbacks=[tf.keras.callbacks.EarlyStopping(patience=5)]\n)\n\n# Retrieve the best model\nbest_hps = tuner.get_best_hyperparameters(num_trials=1)[0]\nbest_model = tuner.get_best_models(num_models=1)[0]\nbest_model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-05T18:34:00.411684Z","iopub.execute_input":"2025-03-05T18:34:00.411998Z","iopub.status.idle":"2025-03-05T18:36:26.210987Z","shell.execute_reply.started":"2025-03-05T18:34:00.411975Z","shell.execute_reply":"2025-03-05T18:36:26.210325Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Ensemble Training Pipeline","metadata":{"_uuid":"f54466a2-92c4-4d39-9c65-8d8861462b93","_cell_guid":"e408f047-a6bc-4c45-9198-3807ba4e0ab1","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# temporal_model = build_temporal_model((1, X_train_scaled.shape[1]))\nhistory = best_model.fit(\n    X_train_3d, y_train,\n    validation_data=(X_val_3d, y_val),\n    epochs=30,  # Reduced epochs\n    batch_size=64,  # Increased batch size\n    callbacks=[\n        EarlyStopping(patience=3, restore_best_weights=True),\n        LearningRateScheduler(lambda epoch: 0.001 * (0.95 ** epoch))\n    ]\n)","metadata":{"_uuid":"19c917e5-1274-4740-bf5d-ea95a46ec251","_cell_guid":"edc56eea-dcad-47a0-adb6-80e8a48ee040","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:37:52.128910Z","iopub.execute_input":"2025-03-05T18:37:52.129213Z","iopub.status.idle":"2025-03-05T18:37:55.344143Z","shell.execute_reply.started":"2025-03-05T18:37:52.129189Z","shell.execute_reply":"2025-03-05T18:37:55.343467Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Ensemble Modeling","metadata":{"_uuid":"136b0ba4-5caa-47a3-a09b-b2101fdadba2","_cell_guid":"bcfebe30-b658-4073-999b-d42e17ed158b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Generate temporal model predictions\ntrain_probs = best_model.predict(X_train_3d).flatten()\nval_probs = best_model.predict(X_val_3d).flatten()\n\n# Create ensemble dataset\nX_train_ensemble = np.column_stack([train_probs, X_train_scaled])\nX_val_ensemble = np.column_stack([val_probs, X_val_scaled])\n\n# Define hyperparameter search space\nparam_dist = {\n    'device': ['cuda'],\n    'tree_method': ['hist'],\n    'learning_rate': uniform(0.01, 0.3),\n    'max_depth': randint(3, 10),\n    'subsample': uniform(0.6, 0.4),  # 0.6-1.0\n    'colsample_bytree': uniform(0.6, 0.4),\n    'gamma': uniform(0, 0.5),\n    'reg_alpha': uniform(0, 1),\n    'reg_lambda': uniform(0, 1),\n    'n_estimators': randint(100, 500)\n}\n\n# Create Bayesian-optimized search\noptimizer = RandomizedSearchCV(\n    estimator=XGBClassifier(\n        objective='binary:logistic',\n        eval_metric='auc',\n        use_label_encoder=False\n        # tree_method='gpu_hist'  # Enable GPU acceleration\n    ),\n    param_distributions=param_dist,\n    n_iter=50,  # Number of parameter combinations\n    scoring='roc_auc',\n    cv=3,\n    n_jobs=-1,\n    verbose=2\n)\n\n# Run optimization\noptimizer.fit(X_train_ensemble, y_train)\n\n# Best model evaluation\nbest_xgb = optimizer.best_estimator_\nbest_xgb.fit(X_train_ensemble, y_train,\n            eval_set=[(X_val_ensemble, y_val)],\n            early_stopping_rounds=20,\n            verbose=False)\n\nensemble_val_probs = best_xgb.predict_proba(X_val_ensemble)[:, 1]\nprint(f\"Optimized AUC: {roc_auc_score(y_val, ensemble_val_probs):.4f}\")\nprint(\"Best parameters:\", optimizer.best_params_)","metadata":{"_uuid":"76f4c906-0edd-4b98-8251-173d25bb9a0f","_cell_guid":"c0637b28-d967-4f0a-a6ac-85051a34b29d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:38:40.049722Z","iopub.execute_input":"2025-03-05T18:38:40.050044Z","iopub.status.idle":"2025-03-05T18:41:49.440908Z","shell.execute_reply.started":"2025-03-05T18:38:40.050021Z","shell.execute_reply":"2025-03-05T18:41:49.440009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Inference & Submission","metadata":{"_uuid":"b578844b-654c-440d-a0fe-a65a6765aa01","_cell_guid":"66f02adf-7411-48b0-8697-521de6a889e5","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Process test set\ntest_df = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\ntest_df['id'] = test_df['id'].apply(lambda x: f\"{int(float(x)):05d}\")\n\ntest_features = []\nfor _, row in tqdm(test_df.iterrows(), desc=\"Processing Test Videos\"):\n    video_path = f\"/kaggle/input/nexar-collision-prediction/test/{row['id']}.mp4\"\n    test_features.append(get_hybrid_features(video_path, 0, 0))  # No event times in test\n\nX_test = scaler.transform(np.array(test_features))\nX_test_3d = X_test.reshape((X_test.shape[0], 1, X_test.shape[1]))\n\n# Generate predictions\ntemporal_probs = best_model.predict(X_test_3d).flatten()\nX_test_ensemble = np.column_stack([temporal_probs, X_test])\n\nfinal_probs = best_xgb.predict_proba(X_test_ensemble)[:, 1]\n\n# Create submission\nsubmission = pd.DataFrame({\n    'id': test_df['id'],\n    'score': final_probs\n})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"\\nSubmission Summary:\")\nprint(submission.describe())","metadata":{"_uuid":"4c022376-b0a2-450e-b572-c8c1d77bf5f1","_cell_guid":"419d7eb8-43aa-461e-8106-6c59a8fd4004","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-03-05T18:43:15.129455Z","iopub.execute_input":"2025-03-05T18:43:15.129781Z","iopub.status.idle":"2025-03-05T18:56:54.551572Z","shell.execute_reply.started":"2025-03-05T18:43:15.129757Z","shell.execute_reply":"2025-03-05T18:56:54.550815Z"}},"outputs":[],"execution_count":null}]}