{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":235484897,"sourceType":"kernelVersion"}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import confusion_matrix, accuracy_score\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport warnings\nfrom scipy.stats import entropy\nimport pickle\nimport gc\n\nwarnings.filterwarnings('ignore')\n\n# Configuration\nTEST_PATH = \"/kaggle/working/\"\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nPREPROCESSED_PATH = \"/kaggle/input/preprocessing/preprocessed/eeg\"\nTRAIN_LABELS_PATH = os.path.join(BASE_PATH, \"train.csv\")\nMODEL_OUTPUT_PATH = os.path.join(TEST_PATH, \"models_rf\")\nFEATURE_CACHE_PATH = os.path.join(TEST_PATH, \"feature_cache\")\n\nos.makedirs(MODEL_OUTPUT_PATH, exist_ok=True)\nos.makedirs(FEATURE_CACHE_PATH, exist_ok=True)\n\nCLASSES = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other']\nN_CLASSES = len(CLASSES)\nBATCH_SIZE = 500  # Process data in batches\n\nprint(\"Starting Memory-Efficient Random Forest EEG Classification\")\n\n# KL Divergence function\ndef kl_divergence_numpy(y_true, y_pred_proba, epsilon=1e-7):\n    y_pred_proba = np.clip(y_pred_proba, epsilon, 1 - epsilon)\n    y_true = np.clip(y_true, epsilon, 1.0)\n    kl_div = np.sum(y_true * np.log(y_true / y_pred_proba), axis=1)\n    return np.mean(kl_div)\n\n# Feature extraction function\ndef extract_features(eeg_data):\n    \"\"\"Extract statistical and frequency domain features from EEG data\"\"\"\n    features = []\n    \n    # Time domain features for each channel\n    for ch in range(eeg_data.shape[0]):  # 19 channels\n        channel_data = eeg_data[ch, :]\n        \n        # Statistical features\n        features.extend([\n            np.mean(channel_data),\n            np.std(channel_data),\n            np.var(channel_data),\n            np.median(channel_data),\n            np.min(channel_data),\n            np.max(channel_data),\n            np.percentile(channel_data, 25),\n            np.percentile(channel_data, 75),\n            np.sum(np.abs(channel_data)),\n            np.sum(channel_data ** 2),\n        ])\n        \n        # Additional statistical measures\n        features.extend([\n            np.abs(np.mean(channel_data)),\n            np.sqrt(np.mean(channel_data ** 2)),\n            entropy(np.abs(channel_data) + 1e-10),\n        ])\n    \n    # Cross-channel features\n    features.extend([\n        np.mean(eeg_data),\n        np.std(eeg_data),\n        np.var(eeg_data),\n        np.corrcoef(eeg_data).mean(),\n    ])\n    \n    return np.array(features)\n\ndef process_eeg_file(eeg_id):\n    \"\"\"Process a single EEG file and return features\"\"\"\n    eeg_path = os.path.join(PREPROCESSED_PATH, f\"{eeg_id}.npy\")\n    try:\n        eeg_data = np.load(eeg_path).astype(np.float32)\n        \n        # Handle NaN/Inf values\n        if np.any(np.isnan(eeg_data)) or np.any(np.isinf(eeg_data)):\n            for ch in range(eeg_data.shape[0]):\n                channel = eeg_data[ch, :]\n                mask = np.isnan(channel) | np.isinf(channel)\n                if np.any(mask):\n                    channel[mask] = np.nanmean(channel)\n        \n        # Standardize each channel\n        eeg_data = (eeg_data - np.mean(eeg_data, axis=1, keepdims=True)) / (np.std(eeg_data, axis=1, keepdims=True) + 1e-7)\n        \n        # Extract features\n        features = extract_features(eeg_data)\n        return features\n        \n    except Exception as e:\n        print(f\"Error loading EEG data for {eeg_id}: {e}\")\n        return None\n\nclass BatchDataGenerator:\n    \"\"\"Generator that yields batches of features and labels\"\"\"\n    def __init__(self, eeg_ids, labels, batch_size=BATCH_SIZE):\n        self.eeg_ids = eeg_ids\n        self.labels = labels\n        self.batch_size = batch_size\n        \n    def __iter__(self):\n        for i in range(0, len(self.eeg_ids), self.batch_size):\n            batch_ids = self.eeg_ids[i:i+self.batch_size]\n            batch_labels = self.labels[i:i+self.batch_size]\n            \n            batch_features = []\n            batch_y = []\n            \n            for eeg_id, label in zip(batch_ids, batch_labels):\n                features = process_eeg_file(eeg_id)\n                if features is not None:\n                    batch_features.append(features)\n                    batch_y.append(label)\n            \n            if batch_features:\n                yield np.array(batch_features), np.array(batch_y)\n\n# Load and preprocess training labels\ntry:\n    train_df = pd.read_csv(TRAIN_LABELS_PATH)\n    print(f\"Loaded {len(train_df)} annotations with {len(train_df['eeg_id'].unique())} unique EEG IDs\")\nexcept Exception as e:\n    print(f\"Error loading training labels: {e}\")\n    exit()\n\n# Encode labels\nlabel_encoder = LabelEncoder()\nlabel_encoder.fit(CLASSES)\ntrain_df['label'] = label_encoder.transform(train_df['expert_consensus'])\n\n# Load data\nprint(\"Loading data...\")\nsuccess_file_path = os.path.join(os.path.dirname(PREPROCESSED_PATH), \"success.csv\")\nif os.path.exists(success_file_path):\n    try:\n        success_df = pd.read_csv(success_file_path)\n        success_ids = set(success_df['eeg_id'].astype(str).tolist())\n        print(f\"Found success file with {len(success_ids)} successful preprocessing entries\")\n    except Exception as e:\n        print(f\"Error loading success file: {e}\")\n        success_ids = set(train_df['eeg_id'].astype(str).tolist())\nelse:\n    print(\"Success file not found, using all available EEG IDs\")\n    success_ids = set(train_df['eeg_id'].astype(str).tolist())\n\n# Filter samples based on success_ids\nvalid_samples = train_df[train_df['eeg_id'].astype(str).isin(success_ids)]\neeg_ids = valid_samples['eeg_id'].astype(str).tolist()\nlabels = valid_samples['label'].values\nprint(f\"Processing {len(eeg_ids)} valid samples\")\n\n# Train-validation split\ntrain_ids, val_ids, train_labels, val_labels = train_test_split(\n    eeg_ids, labels, test_size=0.2, stratify=labels, random_state=42\n)\nprint(f\"Training samples: {len(train_ids)}, Validation samples: {len(val_ids)}\")\n\n# METHOD: Use Random Forest with batch processing and imputation\nprint(\"\\nMethod: Using Random Forest with batch processing and imputation...\")\n\n# Process training data in batches\nprint(\"Processing training data in batches...\")\ntrain_generator = BatchDataGenerator(train_ids, train_labels)\n\n# Collect all training data\nall_train_features = []\nall_train_labels = []\n\nfor batch_X, batch_y in tqdm(train_generator, desc=\"Processing training batches\"):\n    all_train_features.append(batch_X)\n    all_train_labels.append(batch_y)\n    gc.collect()\n\n# Concatenate all batches\nX_train = np.vstack(all_train_features)\ny_train = np.hstack(all_train_labels)\nprint(f\"Training feature matrix shape: {X_train.shape}\")\n\n# Clear intermediate data\ndel all_train_features, all_train_labels\ngc.collect()\n\n# Process validation data\nprint(\"Processing validation data in batches...\")\nval_generator = BatchDataGenerator(val_ids, val_labels)\n\nall_val_features = []\nall_val_labels = []\n\nfor batch_X, batch_y in tqdm(val_generator, desc=\"Processing validation batches\"):\n    all_val_features.append(batch_X)\n    all_val_labels.append(batch_y)\n    gc.collect()\n\nX_val = np.vstack(all_val_features)\ny_val = np.hstack(all_val_labels)\nprint(f\"Validation feature matrix shape: {X_val.shape}\")\n\n# Clear intermediate data\ndel all_val_features, all_val_labels\ngc.collect()\n\n# Impute missing values\nprint(\"Imputing missing values...\")\nimputer = SimpleImputer(strategy='mean')  # Replace NaN with mean of each column\nX_train = imputer.fit_transform(X_train)\nX_val = imputer.transform(X_val)\n\n# Verify no NaN values remain\nif np.any(np.isnan(X_train)) or np.any(np.isnan(X_val)):\n    print(\"Warning: NaN values still present after imputation!\")\nelse:\n    print(\"Imputation successful: No NaN values in training or validation data.\")\n\n# Calculate class weights\nclass_counts = pd.Series(y_train).value_counts().sort_index().values\nclass_weights = len(y_train) / (N_CLASSES * class_counts)\nclass_weight_dict = {i: class_weights[i] for i in range(N_CLASSES)}\n\n# Train Random Forest model\nprint(\"Training Random Forest model...\")\nrf_model = RandomForestClassifier(\n    n_estimators=100,\n    max_depth=10,\n    min_samples_split=5,\n    min_samples_leaf=2,\n    class_weight=class_weight_dict,\n    random_state=42,\n    n_jobs=-1\n)\n\nrf_model.fit(X_train, y_train)\n\n# Make predictions\nprint(\"Making predictions...\")\ntrain_pred_proba = rf_model.predict_proba(X_train)\nval_pred_proba = rf_model.predict_proba(X_val)\n\ntrain_pred = np.argmax(train_pred_proba, axis=1)\nval_pred = np.argmax(val_pred_proba, axis=1)\n\n# Calculate metrics\ntrain_acc = accuracy_score(y_train, train_pred)\nval_acc = accuracy_score(y_val, val_pred)\n\n# Calculate KL divergence\ntrain_true_one_hot = np.eye(N_CLASSES)[y_train.astype(int)]\nval_true_one_hot = np.eye(N_CLASSES)[y_val.astype(int)]\n\ntrain_kl = kl_divergence_numpy(train_true_one_hot, train_pred_proba)\nval_kl = kl_divergence_numpy(val_true_one_hot, val_pred_proba)\n\nprint(f\"\\nFinal Results:\")\nprint(f\"Train Accuracy: {train_acc:.6f}, Train KL Divergence: {train_kl:.6f}\")\nprint(f\"Val Accuracy: {val_acc:.6f}, Val KL Divergence: {val_kl:.6f}\")\n\n# Feature importance plot\nfeature_importance = rf_model.feature_importances_\nplt.figure(figsize=(12, 8))\nsorted_idx = np.argsort(feature_importance)[-20:]  # Top 20 features\nplt.barh(range(len(sorted_idx)), feature_importance[sorted_idx])\nplt.title('Top 20 Feature Importances (Random Forest)')\nplt.xlabel('Importance')\nplt.ylabel('Feature Index')\nplt.tight_layout()\nplt.savefig(os.path.join(MODEL_OUTPUT_PATH, 'feature_importance.png'), dpi=100, bbox_inches='tight')\nplt.close()\n\n# Confusion Matrix\ncm = confusion_matrix(y_val, val_pred)\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=CLASSES, yticklabels=CLASSES)\nplt.title('Confusion Matrix (Validation Set) - Random Forest')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.savefig(os.path.join(MODEL_OUTPUT_PATH, 'confusion_matrix.png'), dpi=100, bbox_inches='tight')\nplt.close()\n\n# Save model\nmodel_path = os.path.join(MODEL_OUTPUT_PATH, \"random_forest_model.pkl\")\nwith open(model_path, 'wb') as f:\n    pickle.dump(rf_model, f)\nprint(f\"Model saved to: {model_path}\")\n\n# Save imputer for future use\nimputer_path = os.path.join(MODEL_OUTPUT_PATH, \"imputer.pkl\")\nwith open(imputer_path, 'wb') as f:\n    pickle.dump(imputer, f)\nprint(f\"Imputer saved to: {imputer_path}\")\n\n# Memory-efficient submission file creation\ndef create_prediction_file_efficient():\n    \"\"\"Create submission file with memory-efficient batch processing\"\"\"\n    test_eeg_path = os.path.join(BASE_PATH, \"test_eegs\")\n    if not os.path.exists(test_eeg_path):\n        print(\"Warning: Test EEG path not found at\", test_eeg_path)\n        print(\"Creating dummy submission\")\n        submission_df = pd.DataFrame({'eeg_id': ['dummy_1', 'dummy_2']})\n        for cls in CLASSES:\n            submission_df[cls] = 1.0 / N_CLASSES\n        submission_path = os.path.join(TEST_PATH, \"submission_rf.csv\")\n        submission_df.to_csv(submission_path, index=False)\n        print(\"Dummy submission file saved to:\", submission_path)\n        return submission_df\n    \n    test_files = [f.replace(\".parquet\", \"\") for f in os.listdir(test_eeg_path) if f.endswith(\".parquet\")]\n    if len(test_files) == 0:\n        print(\"No test files found in\", test_eeg_path)\n        return None\n\n    submission_df = pd.DataFrame({'eeg_id': test_files})\n    for cls in CLASSES:\n        submission_df[cls] = 0.0\n\n    print(f\"Generating predictions for {len(test_files)} test files in batches...\")\n    predictions_made = 0\n    failed_predictions = 0\n    \n    # Process test files in batches\n    total_batches = (len(test_files) + BATCH_SIZE - 1) // BATCH_SIZE\n    \n    for i in tqdm(range(0, len(test_files), BATCH_SIZE), total=total_batches, desc=\"Processing test batches\"):\n        batch_files = test_files[i:i+BATCH_SIZE]\n        batch_features = []\n        valid_files = []\n        \n        # Extract features for current batch\n        for eeg_id in batch_files:\n            features = process_eeg_file(eeg_id)\n            if features is not None:\n                batch_features.append(features)\n                valid_files.append(eeg_id)\n            else:\n                # Set uniform probabilities for failed files\n                uniform_prob = 1.0 / N_CLASSES\n                for cls in CLASSES:\n                    submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = uniform_prob\n                failed_predictions += 1\n        \n        # Make predictions for valid files in the batch\n        if batch_features:\n            try:\n                batch_X = np.array(batch_features)\n                batch_X = imputer.transform(batch_X)  # Apply imputation to test data\n                batch_probs = rf_model.predict_proba(batch_X)\n                \n                # Update submission dataframe\n                for j, eeg_id in enumerate(valid_files):\n                    for k, cls in enumerate(CLASSES):\n                        submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = batch_probs[j, k]\n                    predictions_made += 1\n                \n                # Clean up batch memory\n                del batch_X, batch_probs\n                \n            except Exception as e:\n                print(f\"Error processing batch {i//BATCH_SIZE + 1}: {e}\")\n                # Set uniform probabilities for failed batch\n                uniform_prob = 1.0 / N_CLASSES\n                for eeg_id in valid_files:\n                    for cls in CLASSES:\n                        submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = uniform_prob\n                    failed_predictions += 1\n        \n        # Force garbage collection after each batch\n        gc.collect()\n\n    # Verify all rows have valid probabilities\n    for cls in CLASSES:\n        zero_mask = submission_df[cls] == 0.0\n        if zero_mask.sum() > 0:\n            print(f\"Warning: {zero_mask.sum()} entries have zero probability for class {cls}\")\n            submission_df.loc[zero_mask, cls] = 1.0 / N_CLASSES\n\n    # Normalize probabilities to ensure they sum to 1\n    prob_cols = [cls for cls in CLASSES]\n    row_sums = submission_df[prob_cols].sum(axis=1)\n    for cls in CLASSES:\n        submission_df[cls] = submission_df[cls] / row_sums\n\n    submission_path = os.path.join(TEST_PATH, \"submission_rf.csv\")\n    submission_df.to_csv(submission_path, index=False)\n    \n    print(f\"Submission file saved to: {submission_path}\")\n    print(f\"Successfully made predictions for: {predictions_made}/{len(test_files)} test files\")\n    print(f\"Failed predictions (using uniform distribution): {failed_predictions}/{len(test_files)}\")\n    \n    # Display sample of submission\n    print(\"\\nSample submission entries:\")\n    print(submission_df.head())\n    \n    # Verify submission format\n    print(f\"\\nSubmission shape: {submission_df.shape}\")\n    print(f\"Required columns: {['eeg_id'] + CLASSES}\")\n    print(f\"Actual columns: {list(submission_df.columns)}\")\n    \n    return submission_df\n\ndef validate_submission(submission_df):\n    \"\"\"Validate the submission file format and content\"\"\"\n    if submission_df is None:\n        return False\n    \n    # Check required columns\n    required_cols = ['eeg_id'] + CLASSES\n    if not all(col in submission_df.columns for col in required_cols):\n        print(\"Error: Missing required columns in submission\")\n        return False\n    \n    # Check for NaN values\n    if submission_df.isnull().any().any():\n        print(\"Error: NaN values found in submission\")\n        return False\n    \n    # Check probability constraints\n    prob_cols = CLASSES\n    row_sums = submission_df[prob_cols].sum(axis=1)\n    \n    if not np.allclose(row_sums, 1.0, atol=1e-6):\n        print(\"Warning: Probabilities don't sum to 1.0 for all rows\")\n        print(f\"Row sum range: {row_sums.min():.6f} to {row_sums.max():.6f}\")\n    \n    # Check for negative probabilities\n    if (submission_df[prob_cols] < 0).any().any():\n        print(\"Error: Negative probabilities found\")\n        return False\n    \n    print(\"Submission validation passed!\")\n    return True\n\n# Create and validate submission file\nprint(\"\\n\" + \"=\"*50)\nprint(\"CREATING SUBMISSION FILE\")\nprint(\"=\"*50)\n\ntry:\n    submission_df = create_prediction_file_efficient()\n    \n    if submission_df is not None:\n        is_valid = validate_submission(submission_df)\n        if is_valid:\n            print(\"✓ Submission file created and validated successfully!\")\n        else:\n            print(\"✗ Submission file validation failed!\")\n    else:\n        print(\"✗ Failed to create submission file!\")\n        \nexcept Exception as e:\n    print(f\"Error creating submission file: {e}\")\n    import traceback\n    traceback.print_exc()\n\n# Print final summary\nprint(\"\\n\" + \"=\"*50)\nprint(\"TRAINING SUMMARY\")\nprint(\"=\"*50)\nprint(f\"Model: Random Forest (Memory-Efficient)\")\nprint(f\"Training samples: {len(train_ids)}\")\nprint(f\"Validation samples: {len(val_ids)}\")\nprint(f\"Feature dimensions: {X_train.shape[1] if 'X_train' in locals() else 'N/A'}\")\nprint(f\"Final train accuracy: {train_acc:.6f}\")\nprint(f\"Final validation accuracy: {val_acc:.6f}\")\nprint(f\"Final validation KL divergence: {val_kl:.6f}\")\nprint(\"=\"*50)\n\n# Save label encoder for future use\nlabel_encoder_path = os.path.join(MODEL_OUTPUT_PATH, \"label_encoder.pkl\")\nwith open(label_encoder_path, 'wb') as f:\n    pickle.dump(label_encoder, f)\nprint(f\"Label encoder saved to: {label_encoder_path}\")\n\n# Save training configuration\nconfig = {\n    'classes': CLASSES,\n    'n_classes': N_CLASSES,\n    'batch_size': BATCH_SIZE,\n    'model_params': {\n        'n_estimators': 100,\n        'max_depth': 10,\n        'min_samples_split': 5,\n        'min_samples_leaf': 2,\n        'class_weight': class_weight_dict,\n        'random_state': 42,\n        'n_jobs': -1\n    },\n    'train_accuracy': float(train_acc),\n    'val_accuracy': float(val_acc),\n    'val_kl_divergence': float(val_kl),\n    'n_train_samples': len(train_ids),\n    'n_val_samples': len(val_ids)\n}\n\nconfig_path = os.path.join(MODEL_OUTPUT_PATH, \"training_config.pkl\")\nwith open(config_path, 'wb') as f:\n    pickle.dump(config, f)\nprint(f\"Training configuration saved to: {config_path}\")\n\n# Final cleanup\nprint(\"\\nCleaning up memory...\")\nif 'X_train' in locals():\n    del X_train\nif 'y_train' in locals():\n    del y_train\nif 'X_val' in locals():\n    del X_val\nif 'y_val' in locals():\n    del y_val\nif 'train_pred_proba' in locals():\n    del train_pred_proba\nif 'val_pred_proba' in locals():\n    del val_pred_proba\n\ngc.collect()\nprint(\"Memory cleanup completed!\")\nprint(\"\\n🎉 Memory-Efficient Random Forest training completed successfully! 🎉\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T14:23:44.458119Z","iopub.execute_input":"2025-06-02T14:23:44.458469Z"}},"outputs":[],"execution_count":null}]}