{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":235484897,"sourceType":"kernelVersion"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nfrom sklearn.impute import SimpleImputer\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport warnings\nfrom scipy.stats import entropy\nimport pickle\nimport gc\n\nwarnings.filterwarnings('ignore')\n\n# Configuration\nTEST_PATH = \"/kaggle/working/\"\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nPREPROCESSED_PATH = \"/kaggle/input/preprocessing/preprocessed/eeg\"\nTRAIN_LABELS_PATH = os.path.join(BASE_PATH, \"train.csv\")\nMODEL_OUTPUT_PATH = os.path.join(TEST_PATH, \"models_svm\")\nFEATURE_CACHE_PATH = os.path.join(TEST_PATH, \"feature_cache\")\n\nos.makedirs(MODEL_OUTPUT_PATH, exist_ok=True)\nos.makedirs(FEATURE_CACHE_PATH, exist_ok=True)\n\nCLASSES = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other']\nN_CLASSES = len(CLASSES)\nBATCH_SIZE = 500  # Process data in batches\nN_EPOCHS = 10  # Number of epochs for iterative SVM training\n\nprint(\"Starting Memory-Efficient SVM EEG Classification\")\n\n# KL Divergence function\ndef kl_divergence_numpy(y_true, y_pred_proba, epsilon=1e-7):\n    y_pred_proba = np.clip(y_pred_proba, epsilon, 1 - epsilon)\n    y_true = np.clip(y_true, epsilon, 1.0)\n    kl_div = np.sum(y_true * np.log(y_true / y_pred_proba), axis=1)\n    return np.mean(kl_div)\n\n# Feature extraction function\ndef extract_features(eeg_data):\n    \"\"\"Extract statistical and frequency domain features from EEG data\"\"\"\n    features = []\n    \n    for ch in range(eeg_data.shape[0]):  # 19 channels\n        channel_data = eeg_data[ch, :]\n        \n        if np.all(channel_data == 0) or np.all(np.isnan(channel_data)) or np.all(np.isinf(channel_data)):\n            channel_data = np.random.normal(0, 1e-6, channel_data.shape)\n        \n        features.extend([\n            np.mean(channel_data),\n            np.std(channel_data) if np.std(channel_data) > 0 else 1e-6,\n            np.var(channel_data) if np.var(channel_data) > 0 else 1e-6,\n            np.median(channel_data),\n            np.min(channel_data),\n            np.max(channel_data),\n            np.percentile(channel_data, 25),\n            np.percentile(channel_data, 75),\n            np.sum(np.abs(channel_data)),\n            np.sum(channel_data ** 2),\n        ])\n        \n        features.extend([\n            np.abs(np.mean(channel_data)),\n            np.sqrt(np.mean(channel_data ** 2)),\n            entropy(np.abs(channel_data) + 1e-10) if np.any(channel_data != 0) else 0.0,\n        ])\n    \n    eeg_std = np.std(eeg_data) if np.std(eeg_data) > 0 else 1e-6\n    features.extend([\n        np.mean(eeg_data),\n        eeg_std,\n        np.var(eeg_data) if eeg_std > 0 else 1e-6,\n        np.corrcoef(eeg_data).mean() if not np.all(eeg_data == 0) and eeg_std > 0 else 0.0,\n    ])\n    \n    features = np.array(features)\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n    \n    return features\n\ndef process_eeg_file(eeg_id):\n    \"\"\"Process a single EEG file and return features\"\"\"\n    eeg_path = os.path.join(PREPROCESSED_PATH, f\"{eeg_id}.npy\")\n    try:\n        eeg_data = np.load(eeg_path).astype(np.float32)\n        \n        if np.any(np.isnan(eeg_data)) or np.any(np.isinf(eeg_data)):\n            for ch in range(eeg_data.shape[0]):\n                channel = eeg_data[ch, :]\n                mask = np.isnan(channel) | np.isinf(channel)\n                if np.any(mask):\n                    channel[mask] = np.nanmean(channel) if not np.all(mask) else 0.0\n        \n        std = np.std(eeg_data, axis=1, keepdims=True)\n        mean = np.mean(eeg_data, axis=1, keepdims=True)\n        if np.any(std == 0):\n            eeg_data = np.where(std == 0, np.random.normal(0, 1e-6, eeg_data.shape), eeg_data)\n            std = np.std(eeg_data, axis=1, keepdims=True)\n        \n        eeg_data = (eeg_data - mean) / (std + 1e-7)\n        \n        if np.any(np.isnan(eeg_data)) or np.any(np.isinf(eeg_data)):\n            return None\n        \n        features = extract_features(eeg_data)\n        \n        if np.any(np.isnan(features)) or np.any(np.isinf(features)):\n            return None\n        \n        return features\n        \n    except:\n        return None\n\nclass BatchDataGenerator:\n    \"\"\"Generator that yields batches of features and labels\"\"\"\n    def __init__(self, eeg_ids, labels, batch_size=BATCH_SIZE):\n        self.eeg_ids = eeg_ids\n        self.labels = labels\n        self.batch_size = batch_size\n        \n    def __iter__(self):\n        for i in range(0, len(self.eeg_ids), self.batch_size):\n            batch_ids = self.eeg_ids[i:i+self.batch_size]\n            batch_labels = self.labels[i:i+self.batch_size]\n            \n            batch_features = []\n            batch_y = []\n            \n            for eeg_id, label in zip(batch_ids, batch_labels):\n                features = process_eeg_file(eeg_id)\n                if features is not None:\n                    batch_features.append(features)\n                    batch_y.append(label)\n            \n            if batch_features:\n                yield np.array(batch_features), np.array(batch_y)\n\n# Load and preprocess training labels\ntry:\n    train_df = pd.read_csv(TRAIN_LABELS_PATH)\n    print(f\"Loaded {len(train_df)} annotations with {len(train_df['eeg_id'].unique())} unique EEG IDs\")\nexcept:\n    exit()\n\n# Encode labels\nlabel_encoder = LabelEncoder()\nlabel_encoder.fit(CLASSES)\ntrain_df['label'] = label_encoder.transform(train_df['expert_consensus'])\n\n# Load data\nprint(\"Loading data...\")\nsuccess_file_path = os.path.join(os.path.dirname(PREPROCESSED_PATH), \"success.csv\")\nif os.path.exists(success_file_path):\n    success_df = pd.read_csv(success_file_path)\n    success_ids = set(success_df['eeg_id'].astype(str).tolist())\n    print(f\"Found success file with {len(success_ids)} successful preprocessing entries\")\nelse:\n    success_ids = set(train_df['eeg_id'].astype(str).tolist())\n    print(\"Success file not found, using all available EEG IDs\")\n\n# Filter samples based on success_ids\nvalid_samples = train_df[train_df['eeg_id'].astype(str).isin(success_ids)]\neeg_ids = valid_samples['eeg_id'].astype(str).tolist()\nlabels = valid_samples['label'].values\nprint(f\"Processing {len(eeg_ids)} valid samples\")\n\n# Train-validation split\ntrain_ids, val_ids, train_labels, val_labels = train_test_split(\n    eeg_ids, labels, test_size=0.2, stratify=labels, random_state=42\n)\nprint(f\"Training samples: {len(train_ids)}, Validation samples: {len(val_ids)}\")\n\n# METHOD: Use SVM with iterative training\nprint(\"\\nMethod: Using SVM with iterative training over epochs...\")\n\n# Process training data in batches\nprint(\"Processing training data in batches...\")\ntrain_generator = BatchDataGenerator(train_ids, train_labels)\n\nall_train_features = []\nall_train_labels = []\n\nfor batch_X, batch_y in tqdm(train_generator, desc=\"Processing training batches\"):\n    all_train_features.append(batch_X)\n    all_train_labels.append(batch_y)\n    gc.collect()\n\nX_train = np.vstack(all_train_features)\ny_train = np.hstack(all_train_labels)\nprint(f\"Training feature matrix shape: {X_train.shape}\")\n\n# Process validation data\nprint(\"Processing validation data in batches...\")\nval_generator = BatchDataGenerator(val_ids, val_labels)\n\nall_val_features = []\nall_val_labels = []\n\nfor batch_X, batch_y in tqdm(val_generator, desc=\"Processing validation batches\"):\n    all_val_features.append(batch_X)\n    all_val_labels.append(batch_y)\n    gc.collect()\n\nX_val = np.vstack(all_val_features)\ny_val = np.hstack(all_val_labels)\nprint(f\"Validation feature matrix shape: {X_val.shape}\")\n\n# Impute NaN values\nprint(\"Imputing NaN values in training and validation data...\")\nimputer = SimpleImputer(strategy='mean')\nX_train = imputer.fit_transform(X_train)\nX_val = imputer.transform(X_val)\n\n# Calculate class weights\nclass_counts = pd.Series(y_train).value_counts().sort_index().values\nclass_weights = len(y_train) / (N_CLASSES * class_counts)\nclass_weight_dict = {i: class_weights[i] for i in range(N_CLASSES)}\n\n# Initialize SVM model\nprint(\"Initializing SVM model...\")\nsvm_model = SVC(\n    kernel='rbf',\n    probability=True,\n    class_weight=class_weight_dict,\n    random_state=42,\n    verbose=True\n)\n\n# Iterative training over epochs\nprint(f\"Training SVM model over {N_EPOCHS} epochs...\")\nfor epoch in range(N_EPOCHS):\n    print(f\"Epoch {epoch + 1}/{N_EPOCHS}\")\n    train_generator = BatchDataGenerator(train_ids, train_labels)\n    \n    for batch_X, batch_y in tqdm(train_generator, desc=f\"Training batch (Epoch {epoch + 1})\"):\n        batch_X = imputer.transform(batch_X)\n        svm_model.fit(batch_X, batch_y)\n        gc.collect()\n    \n    train_pred = svm_model.predict(X_train)\n    val_pred = svm_model.predict(X_val)\n    \n    train_pred_proba = svm_model.predict_proba(X_train)\n    val_pred_proba = svm_model.predict_proba(X_val)\n    \n    train_acc = accuracy_score(y_train, train_pred)\n    val_acc = accuracy_score(y_val, val_pred)\n    \n    train_true_one_hot = np.eye(N_CLASSES)[y_train.astype(int)]\n    val_true_one_hot = np.eye(N_CLASSES)[y_val.astype(int)]\n    \n    train_kl = kl_divergence_numpy(train_true_one_hot, train_pred_proba)\n    val_kl = kl_divergence_numpy(val_true_one_hot, val_pred_proba)\n    \n    print(f\"Epoch {epoch + 1} - Train Accuracy: {train_acc:.6f}, Train KL Divergence: {train_kl:.6f}\")\n    print(f\"Epoch {epoch + 1} - Val Accuracy: {val_acc:.6f}, Val KL Divergence: {val_kl:.6f}\")\n\n# Final predictions\nprint(\"Making final predictions...\")\ntrain_pred = svm_model.predict(X_train)\nval_pred = svm_model.predict(X_val)\n\ntrain_pred_proba = svm_model.predict_proba(X_train)\nval_pred_proba = svm_model.predict_proba(X_val)\n\n# Calculate final metrics\ntrain_acc = accuracy_score(y_train, train_pred)\nval_acc = accuracy_score(y_val, val_pred)\n\ntrain_true_one_hot = np.eye(N_CLASSES)[y_train.astype(int)]\nval_true_one_hot = np.eye(N_CLASSES)[y_val.astype(int)]\n\ntrain_kl = kl_divergence_numpy(train_true_one_hot, train_pred_proba)\nval_kl = kl_divergence_numpy(val_true_one_hot, val_pred_proba)\n\nprint(f\"\\nFinal Results:\")\nprint(f\"Train Accuracy: {train_acc:.6f}, Train KL Divergence: {train_kl:.6f}\")\nprint(f\"Val Accuracy: {val_acc:.6f}, Val KL Divergence: {val_kl:.6f}\")\n\n# Feature importance for SVM (using absolute coefficients from linear kernel approximation)\ntry:\n    svm_linear = SVC(kernel='linear', random_state=42)\n    svm_linear.fit(X_train, y_train)\n    \n    feature_importance = np.abs(svm_linear.coef_).mean(axis=0)\n    \n    plt.figure(figsize=(12, 8))\n    sorted_idx = np.argsort(feature_importance)[-20:]\n    plt.barh(range(len(sorted_idx)), feature_importance[sorted_idx])\n    plt.title('Top 20 Feature Importances (SVM - Linear Approximation)')\n    plt.xlabel('Importance')\n    plt.ylabel('Feature Index')\n    plt.tight_layout()\n    plt.savefig(os.path.join(MODEL_OUTPUT_PATH, 'feature_importance.png'), dpi=100, bbox_inches='tight')\n    plt.close()\nexcept:\n    pass\n\n# Confusion Matrix\ncm = confusion_matrix(y_val, val_pred)\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=CLASSES, yticklabels=CLASSES)\nplt.title('Confusion Matrix (Validation Set) - SVM')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.savefig(os.path.join(MODEL_OUTPUT_PATH, 'confusion_matrix.png'), dpi=100, bbox_inches='tight')\nplt.close()\n\n# Save model\nmodel_path = os.path.join(MODEL_OUTPUT_PATH, \"svm_model.pkl\")\nwith open(model_path, 'wb') as f:\n    pickle.dump(svm_model, f)\nprint(f\"Model saved to: {model_path}\")\n\n# Memory-efficient submission file creation\ndef create_prediction_file_efficient():\n    \"\"\"Create submission file with memory-efficient batch processing\"\"\"\n    test_eeg_path = os.path.join(BASE_PATH, \"test_eegs\")\n    if not os.path.exists(test_eeg_path):\n        submission_df = pd.DataFrame({'eeg_id': ['dummy_1', 'dummy_2']})\n        for cls in CLASSES:\n            submission_df[cls] = 1.0 / N_CLASSES\n        submission_path = os.path.join(TEST_PATH, \"submission_svm.csv\")\n        submission_df.to_csv(submission_path, index=False)\n        print(\"Dummy submission file saved to:\", submission_path)\n        return submission_df\n    \n    test_files = [f.replace(\".parquet\", \"\") for f in os.listdir(test_eeg_path) if f.endswith(\".parquet\")]\n    if len(test_files) == 0:\n        return None\n\n    submission_df = pd.DataFrame({'eeg_id': test_files})\n    for cls in CLASSES:\n        submission_df[cls] = 0.0\n\n    print(f\"Generating predictions for {len(test_files)} test files in batches...\")\n    predictions_made = 0\n    failed_predictions = 0\n    \n    total_batches = (len(test_files) + BATCH_SIZE - 1) // BATCH_SIZE\n    \n    for i in tqdm(range(0, len(test_files), BATCH_SIZE), total=total_batches, desc=\"Processing test batches\"):\n        batch_files = test_files[i:i+BATCH_SIZE]\n        batch_features = []\n        valid_files = []\n        \n        for eeg_id in batch_files:\n            features = process_eeg_file(eeg_id)\n            if features is not None:\n                batch_features.append(features)\n                valid_files.append(eeg_id)\n            else:\n                uniform_prob = 1.0 / N_CLASSES\n                for cls in CLASSES:\n                    submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = uniform_prob\n                failed_predictions += 1\n        \n        if batch_features:\n            try:\n                batch_X = np.array(batch_features)\n                batch_X = imputer.transform(batch_X)\n                batch_probs = svm_model.predict_proba(batch_X)\n                \n                for j, eeg_id in enumerate(valid_files):\n                    for k, cls in enumerate(CLASSES):\n                        submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = batch_probs[j, k]\n                    predictions_made += 1\n                \n                del batch_X, batch_probs\n                \n            except:\n                uniform_prob = 1.0 / N_CLASSES\n                for eeg_id in valid_files:\n                    for cls in CLASSES:\n                        submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = uniform_prob\n                    failed_predictions += 1\n        \n        gc.collect()\n\n    for cls in CLASSES:\n        zero_mask = submission_df[cls] == 0.0\n        if zero_mask.sum() > 0:\n            submission_df.loc[zero_mask, cls] = 1.0 / N_CLASSES\n\n    prob_cols = [cls for cls in CLASSES]\n    row_sums = submission_df[prob_cols].sum(axis=1)\n    for cls in CLASSES:\n        submission_df[cls] = submission_df[cls] / row_sums\n\n    submission_path = os.path.join(TEST_PATH, \"submission_svm.csv\")\n    submission_df.to_csv(submission_path, index=False)\n    \n    print(f\"Submission file saved to: {submission_path}\")\n    print(f\"Successfully made predictions for: {predictions_made}/{len(test_files)} test files\")\n    print(f\"Failed predictions (using uniform distribution): {failed_predictions}/{len(test_files)}\")\n    \n    print(\"\\nSample submission entries:\")\n    print(submission_df.head())\n    \n    print(f\"\\nSubmission shape: {submission_df.shape}\")\n    print(f\"Required columns: {['eeg_id'] + CLASSES}\")\n    print(f\"Actual columns: {list(submission_df.columns)}\")\n    \n    return submission_df\n\ndef validate_submission(submission_df):\n    \"\"\"Validate the submission file format and content\"\"\"\n    if submission_df is None:\n        return False\n    \n    required_cols = ['eeg_id'] + CLASSES\n    if not all(col in submission_df.columns for col in required_cols):\n        return False\n    \n    if submission_df.isnull().any().any():\n        return False\n    \n    prob_cols = CLASSES\n    row_sums = submission_df[prob_cols].sum(axis=1)\n    \n    if not np.allclose(row_sums, 1.0, atol=1e-6):\n        return False\n    \n    if (submission_df[prob_cols] < 0).any().any():\n        return False\n    \n    print(\"Submission validation passed!\")\n    return True\n\n# Create and validate submission file\nprint(\"\\n\" + \"=\"*50)\nprint(\"CREATING SUBMISSION FILE\")\nprint(\"=\"*50)\n\ntry:\n    submission_df = create_prediction_file_efficient()\n    \n    if submission_df is not None:\n        is_valid = validate_submission(submission_df)\n        if is_valid:\n            print(\"✓ Submission file created and validated successfully!\")\n        else:\n            print(\"✗ Submission file validation failed!\")\n    else:\n        print(\"✗ Failed to create submission file!\")\n        \nexcept:\n    pass\n\n# Print final summary\nprint(\"\\n\" + \"=\"*50)\nprint(\"TRAINING SUMMARY\")\nprint(\"=\"*50)\nprint(f\"Model: SVM (Memory-Efficient)\")\nprint(f\"Training samples: {len(train_ids)}\")\nprint(f\"Validation samples: {len(val_ids)}\")\nprint(f\"Feature dimensions: {X_train.shape[1]}\")\nprint(f\"Final train accuracy: {train_acc:.6f}\")\nprint(f\"Final validation accuracy: {val_acc:.6f}\")\nprint(f\"Final validation KL divergence: {val_kl:.6f}\")\nprint(\"=\"*50)\n\n# Save label encoder for future use\nlabel_encoder_path = os.path.join(MODEL_OUTPUT_PATH, \"label_encoder.pkl\")\nwith open(label_encoder_path, 'wb') as f:\n    pickle.dump(label_encoder, f)\nprint(f\"Label encoder saved to: {label_encoder_path}\")\n\n# Save training configuration\nconfig = {\n    'classes': CLASSES,\n    'n_classes': N_CLASSES,\n    'batch_size': BATCH_SIZE,\n    'n_epochs': N_EPOCHS,\n    'feature_dimensions': X_train.shape[1],\n    'train_accuracy': float(train_acc),\n    'val_accuracy': float(val_acc),\n    'val_kl_divergence': float(val_kl),\n    'n_train_samples': len(train_ids),\n    'n_val_samples': len(val_ids)\n}\n\nconfig_path = os.path.join(MODEL_OUTPUT_PATH, \"training_config.pkl\")\nwith open(config_path, 'wb') as f:\n    pickle.dump(config, f)\nprint(f\"Training configuration saved to: {config_path}\")\n\n# Final cleanup\nprint(\"\\nCleaning up memory...\")\ngc.collect()\nprint(\"Memory cleanup completed!\")\nprint(\"\\n🎉 Memory-Efficient SVM training completed successfully! 🎉\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}