{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92399,"databundleVersionId":11038207,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport numpy as np\nimport pandas as pd\nimport os\nimport cv2\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Bidirectional, LSTM, Dense\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# Set data directory and load CSV files\nDATA_DIR = \"/kaggle/input/nexar-collision-prediction\"\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\ntest_df  = pd.read_csv(os.path.join(DATA_DIR, \"test.csv\"))\n\nprint(f\"Training videos: {len(train_df)}, Test videos: {len(test_df)}\")\nprint(train_df.head(3))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:10.052056Z","iopub.execute_input":"2025-04-28T17:11:10.054702Z","iopub.status.idle":"2025-04-28T17:11:26.207556Z","shell.execute_reply.started":"2025-04-28T17:11:10.054662Z","shell.execute_reply":"2025-04-28T17:11:26.206621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare video filename columns (zero-pad id to match actual file names, e.g., 0001.mp4)\ntrain_df['filename'] = train_df['id'].apply(lambda x: f\"{int(x):04d}.mp4\")\ntest_df['filename']  = test_df['id'].apply(lambda x: f\"{int(x):04d}.mp4\")\n\n# Check class distribution\nprint(train_df['target'].value_counts())\ntrain_df.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:26.209271Z","iopub.execute_input":"2025-04-28T17:11:26.209672Z","iopub.status.idle":"2025-04-28T17:11:26.244392Z","shell.execute_reply.started":"2025-04-28T17:11:26.209651Z","shell.execute_reply":"2025-04-28T17:11:26.243610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Stratified split into training and validation sets (e.g., 80% train, 20% val)\ntrain_df, val_df = train_test_split(train_df, test_size=0.2, stratify=train_df['target'], random_state=42)\nprint(f\"Train split: {len(train_df)} videos, Validation split: {len(val_df)} videos\")\nprint(\"Class balance in train ->\", train_df['target'].mean())\nprint(\"Class balance in val   ->\", val_df['target'].mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:26.245242Z","iopub.execute_input":"2025-04-28T17:11:26.245495Z","iopub.status.idle":"2025-04-28T17:11:26.758043Z","shell.execute_reply.started":"2025-04-28T17:11:26.245456Z","shell.execute_reply":"2025-04-28T17:11:26.757090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize ResNet50 for feature extraction (ImageNet weights, output = 2048-d vector per frame)\nbase_cnn = ResNet50(weights='imagenet', include_top=False, pooling='avg')\nbase_cnn.trainable = False  # freeze CNN weights\n\n# Helper function to sample N frames uniformly from a video\ndef sample_uniform_frames(video_path, num_frames=15):\n    frames = []\n    cap = cv2.VideoCapture(video_path)\n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if frame_count <= 0:\n        cap.release()\n        return frames  # edge case: no frames\n    # Compute frame indices to sample\n    indices = np.linspace(0, frame_count-1, num=num_frames, dtype=np.int)\n    for idx in indices:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n        ret, frame = cap.read()\n        if not ret:\n            break\n        # Resize frame to 224x224 and convert BGR to RGB for keras preprocess\n        frame = cv2.resize(frame, (224, 224))\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        frames.append(frame)\n    cap.release()\n    return np.array(frames)\n\n# Quick test of frame sampling on one video (not printing image to avoid large output)\nsample_vid = train_df.iloc[0]\nframes = sample_uniform_frames(os.path.join(DATA_DIR, \"train\", sample_vid['filename']), num_frames=5)\n\nif len(frames) > 0:\n    print(f\"Sampled {len(frames)} frames of shape {frames[0].shape} from video {sample_vid['id']}\")\nelse:\n    print(f\"No frames extracted for video {sample_vid['id']} (might be a corrupted video).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:26.759110Z","iopub.execute_input":"2025-04-28T17:11:26.759775Z","iopub.status.idle":"2025-04-28T17:11:31.537225Z","shell.execute_reply.started":"2025-04-28T17:11:26.759749Z","shell.execute_reply":"2025-04-28T17:11:31.536378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features for all train and val videos\ndef extract_features_dataframe(df):\n    X_list = []\n    for _, row in df.iterrows():\n        vid_id = row['id']; fname = row['filename']\n        label = row.get('target', None)\n        video_path = os.path.join(DATA_DIR, \"train\", fname)\n        # Sample frames\n        frames = sample_uniform_frames(video_path, num_frames=15)\n        if len(frames) == 0:\n            # If no frames (corrupt video), skip\n            X_list.append(np.zeros((15, 2048), dtype=np.float32))\n            continue\n        # Preprocess frames for ResNet50\n        frames = preprocess_input(frames.astype(np.float32))\n        # Extract CNN features for frames\n        features = base_cnn.predict(frames, batch_size=15, verbose=0)  # shape (n_frames, 2048)\n        X_list.append(features)\n    return np.array(X_list, dtype=np.float32)\n\n# Perform feature extraction for train and validation sets\nX_train = extract_features_dataframe(train_df)\nX_val   = extract_features_dataframe(val_df)\ny_train = train_df['target'].values\ny_val   = val_df['target'].values\n\nprint(\"Feature extraction complete:\")\nprint(f\"X_train shape: {X_train.shape}, X_val shape: {X_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:31.539606Z","iopub.execute_input":"2025-04-28T17:11:31.539892Z","iopub.status.idle":"2025-04-28T17:11:32.843547Z","shell.execute_reply.started":"2025-04-28T17:11:31.539869Z","shell.execute_reply":"2025-04-28T17:11:32.842477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the LSTM model\nmodel = Sequential([\n    Bidirectional(LSTM(128, dropout=0.5, return_sequences=False), input_shape=(15, 2048)),\n    Dense(1, activation='sigmoid')\n])\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:32.844465Z","iopub.execute_input":"2025-04-28T17:11:32.844679Z","iopub.status.idle":"2025-04-28T17:11:33.051548Z","shell.execute_reply.started":"2025-04-28T17:11:32.844655Z","shell.execute_reply":"2025-04-28T17:11:33.050708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up early stopping\nearly_stop = EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True, verbose=1)\n\n# Train the LSTM model\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=20,\n    batch_size=32,\n    callbacks=[early_stop],\n    verbose=2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:33.052517Z","iopub.execute_input":"2025-04-28T17:11:33.052870Z","iopub.status.idle":"2025-04-28T17:11:45.635427Z","shell.execute_reply.started":"2025-04-28T17:11:33.052840Z","shell.execute_reply":"2025-04-28T17:11:45.634696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate on validation set\nval_loss, val_acc = model.evaluate(X_val, y_val, verbose=0)\nprint(f\"Validation accuracy: {val_acc:.4f}\")\n# Optional: Check precision/recall for more insight (especially since early prediction might favor precision)\nfrom sklearn.metrics import classification_report\nval_preds = model.predict(X_val)[:, 0]\nval_preds_binary = (val_preds >= 0.5).astype(int)\nprint(classification_report(y_val, val_preds_binary, digits=4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:11:45.636285Z","iopub.execute_input":"2025-04-28T17:11:45.636514Z","iopub.status.idle":"2025-04-28T17:11:46.956228Z","shell.execute_reply.started":"2025-04-28T17:11:45.636495Z","shell.execute_reply":"2025-04-28T17:11:46.955285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features for test videos\ndef extract_features_test(df):\n    X_list = []\n    for _, row in df.iterrows():\n        fname = row['filename']\n        video_path = os.path.join(DATA_DIR, \"test\", fname)\n        frames = sample_uniform_frames(video_path, num_frames=15)\n        if len(frames) == 0:\n            X_list.append(np.zeros((15, 2048), dtype=np.float32))\n            continue\n        frames = preprocess_input(frames.astype(np.float32))\n        features = base_cnn.predict(frames, batch_size=15, verbose=0)\n        X_list.append(features)\n    return np.array(X_list, dtype=np.float32)\n\nX_test = extract_features_test(test_df)\nprint(f\"X_test shape: {X_test.shape}\")\n\n# Predict using the trained LSTM model\ntest_preds = model.predict(X_test)  # shape (n_test, 1)\ntest_preds = test_preds[:, 0]       # flatten to 1D array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:12:03.766771Z","iopub.execute_input":"2025-04-28T17:12:03.767126Z","iopub.status.idle":"2025-04-28T17:12:05.875268Z","shell.execute_reply.started":"2025-04-28T17:12:03.767104Z","shell.execute_reply":"2025-04-28T17:12:05.874579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create submission dataframe\nsubmission = test_df.copy()\nsubmission['target'] = test_preds\nsubmission = submission[['id', 'target']]\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T17:12:11.605205Z","iopub.execute_input":"2025-04-28T17:12:11.605519Z","iopub.status.idle":"2025-04-28T17:12:11.627009Z","shell.execute_reply.started":"2025-04-28T17:12:11.605494Z","shell.execute_reply":"2025-04-28T17:12:11.626165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}