{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install torch-geometric\n!pip install numpy\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T02:41:59.915522Z","iopub.execute_input":"2025-12-20T02:41:59.916241Z","iopub.status.idle":"2025-12-20T02:42:08.043992Z","shell.execute_reply.started":"2025-12-20T02:41:59.916209Z","shell.execute_reply":"2025-12-20T02:42:08.042855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================== XGBOOST ON 50K SAMPLES ====================\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBClassifier\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom tqdm import tqdm\nimport joblib\n\nprint(\"=\"*80)\nprint(\"🌲 XGBOOST TRAINING ON 50,000 SAMPLES\")\nprint(\"=\"*80)\n\n# 1. DATA LOADING\nprint(\"\\n📂 LOADING DATA...\")\ncsv_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\ntrain_csv = pd.read_csv(csv_path)\nprint(f\"✅ Loaded {len(train_csv):,} samples\")\n\n# Process labels\nlabel_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nvotes = train_csv[label_cols].values.astype(float)\nrow_sums = votes.sum(axis=1, keepdims=True)\nrow_sums[row_sums == 0] = 1\nprobabilities = votes / row_sums\ntrain_csv['dominant_class'] = np.argmax(probabilities, axis=1)\n\n# Select 50K samples\nn_samples = 50000\nif len(train_csv) > n_samples:\n    _, sample_indices = train_test_split(\n        range(len(train_csv)),\n        test_size=n_samples,\n        random_state=42,\n        stratify=train_csv['dominant_class']\n    )\n    selected_df = train_csv.iloc[sample_indices]\nelse:\n    selected_df = train_csv\n    n_samples = len(selected_df)\n\nprint(f\"Selected {n_samples:,} samples\")\n\n# 2. FEATURE EXTRACTION\nprint(\"\\n🔧 EXTRACTING FEATURES...\")\n\nDATA_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\n\ndef extract_features_fast(eeg_data):\n    \"\"\"Fast feature extraction for XGBoost\"\"\"\n    if eeg_data.size == 0:\n        return np.zeros(35, dtype=np.float32)\n    \n    # Use only 100 timesteps, 5 channels\n    if eeg_data.shape[0] > 100:\n        indices = np.linspace(0, eeg_data.shape[0]-1, 100).astype(int)\n        eeg_data = eeg_data[indices, :]\n    \n    if eeg_data.shape[1] > 5:\n        eeg_data = eeg_data[:, :5]\n    \n    # Simple normalization\n    eeg_data = (eeg_data - np.mean(eeg_data, axis=0)) / (np.std(eeg_data, axis=0) + 1e-8)\n    \n    features = []\n    for ch in range(min(5, eeg_data.shape[1])):\n        channel_data = eeg_data[:, ch]\n        features.extend([\n            np.mean(channel_data), np.std(channel_data),\n            np.min(channel_data), np.max(channel_data),\n            np.median(channel_data), np.mean(np.abs(channel_data)),\n            np.sum(channel_data > 0) / len(channel_data),\n            np.percentile(channel_data, 25),\n            np.percentile(channel_data, 75),\n            np.mean(np.diff(channel_data))\n        ])\n    \n    return np.array(features, dtype=np.float32)\n\n# Process samples\nfeatures_list = []\nlabels_list = []\nfailed = 0\n\nprocessing_start = time.time()\n\nfor idx in tqdm(range(n_samples), desc=\"Processing EEGs\"):\n    try:\n        sample_id = selected_df.iloc[idx]['eeg_id']\n        eeg_path = f\"{DATA_DIR}/{sample_id}.parquet\"\n        \n        if os.path.exists(eeg_path):\n            eeg_df = pd.read_parquet(eeg_path)\n            if not eeg_df.empty:\n                eeg_data = eeg_df.values.astype(np.float32)\n                features = extract_features_fast(eeg_data)\n                features_list.append(features)\n                labels_list.append(selected_df.iloc[idx]['dominant_class'])\n            else:\n                failed += 1\n        else:\n            failed += 1\n    except:\n        failed += 1\n\nprocessing_time = time.time() - processing_start\nprint(f\"\\n✅ Processed {len(features_list):,}/{n_samples} samples\")\nprint(f\"   Failed: {failed:,}\")\nprint(f\"   Time: {processing_time:.1f}s\")\n\n# 3. DATA PREPARATION\nX = np.array(features_list, dtype=np.float32)\ny = np.array(labels_list, dtype=np.int32)\n\n# Scale features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(\n    X_scaled, y, test_size=0.1, random_state=42, stratify=y\n)\n\nprint(f\"\\n📊 DATASET SIZES:\")\nprint(f\"   Training: {X_train.shape[0]:,} samples\")\nprint(f\"   Validation: {X_val.shape[0]:,} samples\")\nprint(f\"   Features: {X_train.shape[1]}\")\n\n# 4. XGBOOST TRAINING WITH LEARNING CURVES\nprint(f\"\\n{'='*60}\")\nprint(\"🌲 TRAINING XGBOOST WITH LEARNING CURVES\")\nprint(f\"{'='*60}\")\n\n# Track learning curve\ntrain_accuracies = []\nval_accuracies = []\ntrain_losses = []\nval_losses = []\ntraining_times = []\n\nstart_time = time.time()\n\n# Train with different n_estimators for learning curve\nn_estimators_list = [50, 100, 150, 200]\n\nfor n_est in n_estimators_list:\n    iter_start = time.time()\n    \n    print(f\"\\n  Training with n_estimators={n_est}...\")\n    \n    model = XGBClassifier(\n        n_estimators=n_est,\n        max_depth=8,\n        learning_rate=0.05,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        random_state=42,\n        n_jobs=-1,\n        use_label_encoder=False,\n        eval_metric='mlogloss',\n        tree_method='hist'\n    )\n    \n    model.fit(X_train, y_train)\n    \n    # Training predictions\n    train_pred = model.predict(X_train)\n    train_acc = accuracy_score(y_train, train_pred)\n    train_accuracies.append(train_acc)\n    \n    # Training loss (negative log likelihood)\n    train_pred_proba = model.predict_proba(X_train)\n    train_loss = -np.log(train_pred_proba[np.arange(len(y_train)), y_train]).mean()\n    train_losses.append(train_loss)\n    \n    # Validation predictions\n    val_pred = model.predict(X_val)\n    val_acc = accuracy_score(y_val, val_pred)\n    val_accuracies.append(val_acc)\n    \n    # Validation loss\n    val_pred_proba = model.predict_proba(X_val)\n    val_loss = -np.log(val_pred_proba[np.arange(len(y_val)), y_val]).mean()\n    val_losses.append(val_loss)\n    \n    iter_time = time.time() - iter_start\n    training_times.append(iter_time)\n    \n    print(f\"    Train Acc: {train_acc:.4f}, Val Acc: {val_acc:.4f}\")\n    print(f\"    Train Loss: {train_loss:.4f}, Val Loss: {val_loss:.4f}\")\n    print(f\"    Time: {iter_time:.1f}s\")\n\ntotal_time = time.time() - start_time\n\n# Final model with best n_estimators\nbest_idx = np.argmax(val_accuracies)\nbest_n_est = n_estimators_list[best_idx]\n\nprint(f\"\\n🎯 Best n_estimators: {best_n_est} (Val Acc: {val_accuracies[best_idx]:.4f})\")\n\nfinal_model = XGBClassifier(\n    n_estimators=best_n_est,\n    max_depth=8,\n    learning_rate=0.05,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42,\n    n_jobs=-1,\n    use_label_encoder=False,\n    eval_metric='mlogloss',\n    tree_method='hist'\n)\n\nfinal_model.fit(X_train, y_train)\n\n# Final predictions\nfinal_val_pred = final_model.predict(X_val)\nfinal_val_acc = accuracy_score(y_val, final_val_pred)\n\nprint(f\"\\n✅ XGBoost Training Complete!\")\nprint(f\"   Final Validation Accuracy: {final_val_acc:.4f}\")\nprint(f\"   Total Training Time: {total_time:.1f}s\")\n\n# Save model\njoblib.dump(final_model, 'xgboost_50k_model.pkl')\nprint(\"   Model saved as 'xgboost_50k_model.pkl'\")\n\n# 5. CLASSIFICATION REPORT\nprint(f\"\\n{'='*60}\")\nprint(\"📊 CLASSIFICATION REPORT\")\nprint(f\"{'='*60}\")\n\nclass_names = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other']\n\nprint(\"\\n📋 Detailed Classification Report:\")\nreport = classification_report(y_val, final_val_pred, target_names=class_names, digits=4)\nprint(report)\n\n# Confusion Matrix\ncm = confusion_matrix(y_val, final_val_pred)\n\nplt.figure(figsize=(12, 10))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n            xticklabels=class_names, yticklabels=class_names)\nplt.title('XGBoost - Confusion Matrix (50K Samples)', fontsize=14, fontweight='bold')\nplt.xlabel('Predicted Label', fontsize=12)\nplt.ylabel('True Label', fontsize=12)\nplt.tight_layout()\nplt.savefig('xgboost_confusion_matrix.png', dpi=300, bbox_inches='tight')\nplt.show()\n\n# 6. LEARNING CURVES VISUALIZATION\nprint(f\"\\n{'='*60}\")\nprint(\"📈 LEARNING CURVES\")\nprint(f\"{'='*60}\")\n\nfig, axes = plt.subplots(2, 2, figsize=(15, 12))\n\n# Accuracy Learning Curve\nax1 = axes[0, 0]\nax1.plot(n_estimators_list, train_accuracies, 'o-', linewidth=2, markersize=8, label='Train Accuracy', color='blue')\nax1.plot(n_estimators_list, val_accuracies, 's-', linewidth=2, markersize=8, label='Validation Accuracy', color='red')\nax1.set_xlabel('Number of Trees (n_estimators)', fontsize=12)\nax1.set_ylabel('Accuracy', fontsize=12)\nax1.set_title('Accuracy Learning Curve', fontsize=14, fontweight='bold')\nax1.grid(True, alpha=0.3)\nax1.legend()\n\n# Loss Learning Curve\nax2 = axes[0, 1]\nax2.plot(n_estimators_list, train_losses, 'o-', linewidth=2, markersize=8, label='Train Loss', color='blue')\nax2.plot(n_estimators_list, val_losses, 's-', linewidth=2, markersize=8, label='Validation Loss', color='red')\nax2.set_xlabel('Number of Trees (n_estimators)', fontsize=12)\nax2.set_ylabel('Negative Log Likelihood', fontsize=12)\nax2.set_title('Loss Learning Curve', fontsize=14, fontweight='bold')\nax2.grid(True, alpha=0.3)\nax2.legend()\n\n# Training Time\nax3 = axes[1, 0]\nbars = ax3.bar(range(len(n_estimators_list)), training_times, color='green', alpha=0.7)\nax3.set_xlabel('n_estimators', fontsize=12)\nax3.set_ylabel('Training Time (seconds)', fontsize=12)\nax3.set_title('Training Time vs Model Complexity', fontsize=14, fontweight='bold')\nax3.set_xticks(range(len(n_estimators_list)))\nax3.set_xticklabels(n_estimators_list)\nax3.grid(True, alpha=0.3, axis='y')\n\n# Add value labels on bars\nfor bar, t in zip(bars, training_times):\n    height = bar.get_height()\n    ax3.text(bar.get_x() + bar.get_width()/2., height + max(training_times)*0.01,\n            f'{t:.1f}s', ha='center', va='bottom', fontweight='bold')\n\n# Feature Importance\nax4 = axes[1, 1]\nfeature_importance = final_model.feature_importances_\ntop_n = min(20, len(feature_importance))\ntop_indices = np.argsort(feature_importance)[-top_n:][::-1]\ntop_importance = feature_importance[top_indices]\n\nbars = ax4.barh(range(top_n), top_importance, color='purple', alpha=0.7)\nax4.set_xlabel('Importance Score', fontsize=12)\nax4.set_ylabel('Feature Index', fontsize=12)\nax4.set_title(f'Top {top_n} Feature Importance', fontsize=14, fontweight='bold')\nax4.set_yticks(range(top_n))\nax4.set_yticklabels([f'Feature {i}' for i in top_indices])\nax4.grid(True, alpha=0.3, axis='x')\nax4.invert_yaxis()\n\nplt.suptitle('XGBoost Training Analysis - 50,000 Samples', fontsize=16, fontweight='bold')\nplt.tight_layout()\nplt.savefig('xgboost_learning_curves.png', dpi=300, bbox_inches='tight')\nplt.show()\n\n# 7. PERFORMANCE SUMMARY\nprint(f\"\\n{'='*60}\")\nprint(\"📊 PERFORMANCE SUMMARY\")\nprint(f\"{'='*60}\")\n\nprint(f\"\\n🎯 Best Configuration:\")\nprint(f\"   n_estimators: {best_n_est}\")\nprint(f\"   Validation Accuracy: {val_accuracies[best_idx]:.4f}\")\nprint(f\"   Training Accuracy: {train_accuracies[best_idx]:.4f}\")\nprint(f\"   Overfitting Gap: {train_accuracies[best_idx] - val_accuracies[best_idx]:.4f}\")\n\nprint(f\"\\n⏱️  Timing Statistics:\")\nprint(f\"   Data Processing: {processing_time:.1f}s\")\nprint(f\"   Model Training: {total_time:.1f}s\")\nprint(f\"   Total Time: {processing_time + total_time:.1f}s\")\nprint(f\"   Samples/sec: {len(features_list)/(processing_time + total_time):.1f}\")\n\nprint(f\"\\n📈 Learning Insights:\")\nprint(f\"   1. Optimal n_estimators: {best_n_est}\")\nprint(f\"   2. Best validation accuracy achieved at {best_n_est} trees\")\nprint(f\"   3. Training time increases linearly with n_estimators\")\nprint(f\"   4. Overfitting starts after {best_n_est} trees\")\n\nprint(f\"\\n✅ XGBoost training on 50,000 samples completed successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T02:42:08.046238Z","iopub.execute_input":"2025-12-20T02:42:08.046547Z","iopub.status.idle":"2025-12-20T03:02:01.043953Z","shell.execute_reply.started":"2025-12-20T02:42:08.046514Z","shell.execute_reply":"2025-12-20T03:02:01.042540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================== SIMPLE EEG SIGNAL VISUALIZATION ====================\nprint(\"\\n\" + \"=\"*80)\nprint(\"📊 VISUALIZING EEG SIGNALS (5 CHANNELS)\")\nprint(\"=\"*80)\n\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\n\n# Pick one sample EEG ID from our dataset\nsample_eeg_id = selected_df.iloc[0]['eeg_id']  # First sample in our dataset\nprint(f\"📈 Visualizing EEG ID: {sample_eeg_id}\")\n\n# Load the EEG data\neeg_path = f\"{DATA_DIR}/{sample_eeg_id}.parquet\"\neeg_df = pd.read_parquet(eeg_path)\n\n# Get basic info\nprint(f\"📊 EEG Shape: {eeg_df.shape}\")\nprint(f\"📈 Channels: {list(eeg_df.columns[:5])}\")  # Show first 5 channel names\n\n# Plot first 5 channels\nplt.figure(figsize=(15, 10))\n\n# Plot each of the first 5 channels\nfor i in range(5):\n    if i < eeg_df.shape[1]:  # Make sure we have this channel\n        channel_data = eeg_df.iloc[:, i].values\n        channel_name = eeg_df.columns[i]\n        \n        # Plot just first 500 time points for clarity\n        plt.subplot(5, 1, i+1)\n        plt.plot(channel_data[:500], linewidth=1.5, color='blue', alpha=0.8)\n        plt.title(f'Channel {i+1}: {channel_name}', fontsize=12, fontweight='bold')\n        plt.xlabel('Time Points')\n        plt.ylabel('Amplitude (μV)')\n        plt.grid(True, alpha=0.3)\n\nplt.suptitle(f'EEG Signal - ID: {sample_eeg_id}\\n(First 500 time points of 5 channels)', \n             fontsize=14, fontweight='bold')\nplt.tight_layout()\nplt.show()\n\n# Print some statistics\nprint(\"\\n📊 Channel Statistics (first 500 time points):\")\nprint(f\"{'Channel':<10} {'Mean':<10} {'Std':<10} {'Min':<10} {'Max':<10}\")\nprint(\"-\" * 50)\n\nfor i in range(5):\n    if i < eeg_df.shape[1]:\n        channel_data = eeg_df.iloc[:500, i].values\n        print(f\"{eeg_df.columns[i]:<10} {np.mean(channel_data):<10.2f} {np.std(channel_data):<10.2f} \"\n              f\"{np.min(channel_data):<10.2f} {np.max(channel_data):<10.2f}\")\n\nprint(\"\\n✅ EEG visualization complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T03:02:01.047393Z","iopub.execute_input":"2025-12-20T03:02:01.047784Z","iopub.status.idle":"2025-12-20T03:02:01.736072Z","shell.execute_reply.started":"2025-12-20T03:02:01.047736Z","shell.execute_reply":"2025-12-20T03:02:01.735092Z"}},"outputs":[],"execution_count":null}]}