{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dashcam Collision Prediction: Baseline InceptionV3 + RandomForest Pipeline\n\nThis notebook provides a **simple, end‑to‑end baseline** for the Nexar Collision Prediction competition. You can:\n\n1. **Load & preprocess** the train/test CSVs  \n2. **Sample** a fixed number of frames from each video  \n3. **Extract deep features** using a pre‑trained InceptionV3 (imagenet) model  \n4. **Aggregate** those frame features into a single vector per video  \n5. **Standardize** and **split** into train/validation sets  \n6. **Train** a RandomForest classifier on the extracted features  \n7. **Validate** with ROC‑AUC  \n8. **Infer** on the test set and create a `baseline_submission.csv`\n\n---\n\n## Why use this baseline?\n\n-  **Modular**: Easily swap out the CNN (e.g. EfficientNet, ResNet)  \n-  **Extensible**: Add optical‑flow, YOLO region crops, temporal models  \n-  **Strong starting point**: Often beats a “naïve” logistic regression on raw pixels\n\n---\n\n## How to build on it\n\n- **Change the backbone**: Replace InceptionV3 with your favorite model  \n- **Tune your frames**: Increase or decrease `num_frames`, sample around the alert time  \n- **Add more features**: Motion statistics, scene‑change detectors, object crops  \n- **Swap the classifier**: Try XGBoost, LightGBM, or a small MLP  \n- **Cross‑validate**: Use k‑folds for more robust estimates  \n- **Ensemble**: Blend multiple pipelines together\n\nFeel free to fork, adapt, and **share your improvements**! If you find this notebook helpful, please give it an **upvote**. Happy modeling! 😊\n","metadata":{}},{"cell_type":"code","source":"#!/usr/bin/env python3\n# -*- coding: utf-8 -*-\n\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_auc_score\n\nfrom tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\n\n# 📥 1. Load CSVs & pad IDs\ndf       = pd.read_csv('/kaggle/input/nexar-collision-prediction/train.csv')\ndf_test  = pd.read_csv('/kaggle/input/nexar-collision-prediction/test.csv')\ndf['id']       = df['id'].astype(str).str.zfill(5)\ndf_test['id']  = df_test['id'].astype(str).str.zfill(5)\n\n# 📂 2. Define video folders & filenames\ntrain_dir = '/kaggle/input/nexar-collision-prediction/train/'\ntest_dir  = '/kaggle/input/nexar-collision-prediction/test/'\n\ndf['train_videos'] = df['id']       + '.mp4'\ndf_test['test_videos'] = df_test['id'] + '.mp4'\n\n# 🔍 3. Quick sanity prints\nprint(f\"Örnek Train ID:\\n{df['id'].head().to_list()}\")\nprint(f\"Örnek Test ID:\\n{df_test['id'].head().to_list()}\")\nprint(f\"Toplam Train Videosu: {len(df['train_videos'])}\")\nprint(f\"Toplam Test Videosu:  {len(df_test['test_videos'])}\")\n\n# 🖼️ 4. Frame sampling helper\ndef extract_frames(path, num_frames=16, size=(224,224)):\n    cap = cv2.VideoCapture(path)\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if total <= 0:\n        cap.release()\n        return np.empty((0, *size, 3), dtype=np.uint8)\n    step = max(total // num_frames, 1)\n    frames = []\n    for i in range(num_frames):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, i * step)\n        ret, frame = cap.read()\n        if not ret:\n            break\n        frame = cv2.resize(frame, size)\n        frames.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))\n    cap.release()\n    return np.stack(frames) if frames else np.empty((0, *size, 3), dtype=np.uint8)\n\n# 🔧 5. Load CNN for feature extraction\nbase_model = InceptionV3(weights='imagenet', include_top=False, pooling='avg')\nfeat_dim   = base_model.output_shape[-1]\n\ndef get_features(ids, folder):\n    feats = []\n    for vid in tqdm(ids, desc=f\"Extracting from {folder}\"):\n        path = os.path.join(folder, f\"{vid}.mp4\")\n        frames = extract_frames(path)\n        if frames.size == 0:\n            feats.append(np.zeros(feat_dim, dtype=np.float32))\n            continue\n        x = preprocess_input(frames.astype('float32'))\n        f = base_model.predict(x, batch_size=16, verbose=0)\n        feats.append(f.mean(axis=0))\n    return np.vstack(feats)\n\n# ⚙️ 6. Extract features\nX_train_full = get_features(df['id'],      train_dir)\nX_test       = get_features(df_test['id'], test_dir)\n\n# 🔄 7. Scale & split\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_train_full)\ny        = df['target'].values\n\nX_tr, X_val, y_tr, y_val = train_test_split(\n    X_scaled, y,\n    test_size=0.2,\n    stratify=y,\n    random_state=42\n)\n\n# 🌲 8. Train RandomForest baseline\nclf = RandomForestClassifier(n_estimators=100, n_jobs=-1, random_state=42)\nclf.fit(X_tr, y_tr)\n\n# 📈 9. Validate\nval_pred = clf.predict_proba(X_val)[:,1]\nprint(\"Validation ROC‑AUC:\", roc_auc_score(y_val, val_pred))\n\n# 💾 10. Inference & submission\nX_test_scaled = scaler.transform(X_test)\ntest_pred     = clf.predict_proba(X_test_scaled)[:,1]\n\nsubmission = pd.DataFrame({\n    'id':    df_test['id'],\n    'score': test_pred\n})\nsubmission.to_csv('baseline_submission.csv', index=False)\nprint(\"✅ Written → baseline_submission.csv\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}