{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":12324037,"sourceType":"datasetVersion","datasetId":7768286}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required packages\n!pip install numpy pandas scikit-learn scipy psutil\n!pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118\n!pip install gspread google-auth google-auth-oauthlib google-auth-httplib2\n!pip install pyarrow  # For parquet file support\n!pip install gputil","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python3\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset, Subset\nfrom torch.cuda.amp import GradScaler, autocast\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, QuantileTransformer, PowerTransformer, MinMaxScaler\nfrom sklearn.model_selection import KFold, TimeSeriesSplit, StratifiedKFold, GroupKFold\nfrom sklearn.decomposition import PCA, TruncatedSVD, FastICA\nfrom sklearn.cluster import KMeans, DBSCAN\nfrom sklearn.ensemble import RandomForestRegressor, IsolationForest\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom scipy.stats import pearsonr, spearmanr, rankdata, zscore\nfrom scipy.special import expit, logit\nimport warnings\nimport gc\nimport psutil\nimport time\nfrom typing import List, Dict, Tuple, Optional, Set, Any, Union\nfrom collections import defaultdict\nimport random\nimport json\nimport hashlib\nfrom datetime import datetime, timedelta\nimport os\nimport itertools\nfrom dataclasses import dataclass, field\nimport traceback\nimport signal\nimport sys\nfrom concurrent.futures import ThreadPoolExecutor, TimeoutError\nimport threading\nimport math\nimport pickle\nimport socket\nimport uuid\nfrom functools import partial\n\ntry:\n    import gspread\n    from google.oauth2.service_account import Credentials\n    GSPREAD_AVAILABLE = True\nexcept ImportError:\n    GSPREAD_AVAILABLE = False\n    print(\"Warning: Google Sheets libraries not available. Install with: pip install gspread google-auth\")\n\nwarnings.filterwarnings('ignore')\n\nos.environ['CUDA_VISIBLE_DEVICES'] = '0'\n\ndef to_python_type(obj):\n    if isinstance(obj, np.integer):\n        return int(obj)\n    elif isinstance(obj, np.floating):\n        return float(obj)\n    elif isinstance(obj, np.ndarray):\n        return obj.tolist()\n    elif torch is not None and isinstance(obj, torch.Tensor):\n        return obj.detach().cpu().numpy().tolist()\n    elif isinstance(obj, (np.bool_, bool)):\n        return bool(obj)\n    elif isinstance(obj, list):\n        return [to_python_type(item) for item in obj]\n    elif isinstance(obj, dict):\n        return {key: to_python_type(value) for key, value in obj.items()}\n    else:\n        return obj\n\ndef make_hashable(obj):\n    if isinstance(obj, dict):\n        return tuple(sorted((k, make_hashable(v)) for k, v in obj.items()))\n    elif isinstance(obj, list):\n        return tuple(make_hashable(item) for item in obj)\n    elif isinstance(obj, set):\n        return tuple(sorted(make_hashable(item) for item in obj))\n    elif isinstance(obj, (str, int, float, bool, type(None))):\n        return obj\n    elif isinstance(obj, tuple):\n        return tuple(make_hashable(item) for item in obj)\n    else:\n        return str(obj)\n\n@dataclass\nclass NoiseDetectionConfig:\n    use_isolation_forest: bool = True\n    use_local_outlier_factor: bool = True\n    use_statistical_detection: bool = True\n    use_label_consistency: bool = True\n    \n    isolation_contamination: float = 0.05\n    lof_contamination: float = 0.05\n    lof_novelty: bool = True  # Fixed: Set to True to enable predict()\n    statistical_threshold: float = 3.0\n    label_consistency_window: int = 100\n    label_consistency_threshold: float = 0.1\n    \n    consensus_threshold: float = 0.5\n\n@dataclass\nclass FeatureGroupConfig:\n    market_features: List[str] = field(default_factory=lambda: [\n        \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n    ])\n    \n    microstructure_features: List[str] = field(default_factory=lambda: [\n        \"volume_weighted_sell\", \"buy_sell_ratio\", \n        \"selling_pressure\", \"effective_spread_proxy\",\n        \"bid_ask_imbalance\", \"flow_toxicity\",\n        \"volume_concentration\", \"liquidity_consumption\",\n        \"price_pressure\", \"order_imbalance\", \"relative_spread\",\n        \"liquidity_ratio\", \"trade_intensity\", \"volume_imbalance\",\n        \"price_efficiency\", \"market_depth_ratio\"\n    ])\n    \n    core_proprietary_features: List[str] = field(default_factory=lambda: [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \n        \"X674\", \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \n        \"X612\", \"X888\", \"X421\", \"X333\", \"X292\"\n    ])\n    \n    # Updated dropout rates closer to 0.5\n    dropout_rates: Dict[str, float] = field(default_factory=lambda: {\n        'market': 0.45,\n        'microstructure': 0.5,\n        'core_proprietary': 0.55,\n        'other_proprietary': 0.6,\n        'interactions': 0.5,\n        'engineered': 0.5\n    })\n    \n    noise_levels: Dict[str, float] = field(default_factory=lambda: {\n        'market': 0.005,\n        'microstructure': 0.005,\n        'core_proprietary': 0.005,\n        'other_proprietary': 0.01,\n        'interactions': 0.005,\n        'engineered': 0.005\n    })\n    \n    lr_multipliers: Dict[str, float] = field(default_factory=lambda: {\n        'market': 1.0,\n        'microstructure': 0.8,\n        'core_proprietary': 0.5,\n        'other_proprietary': 0.3,\n        'interactions': 0.6,\n        'engineered': 0.7\n    })\n    \n    def get_feature_indices(self, all_features: List[str]) -> Dict[str, List[int]]:\n        indices = {}\n        \n        indices['market'] = [\n            i for i, f in enumerate(all_features) \n            if f in self.market_features\n        ]\n        \n        indices['microstructure'] = [\n            i for i, f in enumerate(all_features) \n            if f in self.microstructure_features\n        ]\n        \n        indices['core_proprietary'] = [\n            i for i, f in enumerate(all_features) \n            if f in self.core_proprietary_features\n        ]\n        \n        indices['other_proprietary'] = [\n            i for i, f in enumerate(all_features) \n            if f.startswith('X') and f not in self.core_proprietary_features\n        ]\n        \n        indices['engineered'] = [\n            i for i, f in enumerate(all_features)\n            if any(x in f for x in ['_ratio_', '_imbalance', '_pressure', '_concentration'])\n        ]\n        \n        return indices\n\ndef create_advanced_microstructure_features(df: pd.DataFrame) -> pd.DataFrame:\n    df = df.copy()\n    \n    eps = 1e-8\n    \n    df['volume_weighted_sell'] = df['sell_qty'].values * df['volume'].values\n    df['buy_sell_ratio'] = df['buy_qty'].values / (df['sell_qty'].values + eps)\n    df['selling_pressure'] = df['sell_qty'].values / (df['volume'].values + eps)\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'].values - df['sell_qty'].values) / (df['volume'].values + eps)\n    \n    df['bid_ask_imbalance'] = (\n        (df['bid_qty'] - df['ask_qty']) / \n        (df['bid_qty'] + df['ask_qty'] + eps)\n    )\n    \n    df['flow_toxicity'] = df['sell_qty'] / (df['buy_qty'] + eps)\n    \n    df['volume_concentration'] = (\n        (df['buy_qty'] + df['sell_qty']) / (df['volume'] + eps)\n    )\n    \n    df['liquidity_consumption'] = (\n        (df['buy_qty'] + df['sell_qty']) / \n        (df['bid_qty'] + df['ask_qty'] + eps)\n    )\n    \n    df['price_pressure'] = (\n        (df['buy_qty'] - df['sell_qty']) / (df['volume'] + eps)\n    )\n    \n    df['order_imbalance'] = (\n        df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + eps)\n    )\n    \n    df['relative_spread'] = (\n        np.abs(df['buy_qty'] - df['sell_qty']) / \n        (df['bid_qty'] + df['ask_qty'] + eps)\n    )\n    \n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + eps)\n    df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + eps)\n    df['volume_imbalance'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + eps)\n    df['price_efficiency'] = 1 - np.abs(df['order_imbalance'] - 0.5) * 2\n    df['market_depth_ratio'] = df['bid_qty'] / (df['ask_qty'] + eps)\n    \n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    return df\n\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    CREDENTIALS_FILE = '/kaggle/input/greedy3/forward-leaf-464213-u2-489add6f6da1.json'\n    SPREADSHEET_URL = 'https://docs.google.com/spreadsheets/d/1nBW-fJxYwtopFNFuTuqHirBsK9QkjojXJU8DmaB8OnI/edit?gid=0#gid=0'\n    \n    FEATURE_GROUP_CONFIG = FeatureGroupConfig()\n    NOISE_DETECTION_CONFIG = NoiseDetectionConfig()\n    \n    CORE_FEATURES = (\n        FEATURE_GROUP_CONFIG.market_features +\n        FEATURE_GROUP_CONFIG.microstructure_features +\n        FEATURE_GROUP_CONFIG.core_proprietary_features\n    )\n    \n    MAX_BATCH_SIZE_PREDICT = 10000\n    CLEAR_MEMORY_INTERVAL = 5\n    USE_MIXED_PRECISION = True\n    \n    USE_MULTI_GPU = False\n    DEVICE_IDS = [0]\n    \n    BASELINE_LAYERS = [256, 128, 64, 1]\n    BASELINE_DROPOUT = 0.5\n    BASELINE_NOISE = 0.005\n    \n    MAX_RUNTIME_HOURS = 8.0\n    CHECKPOINT_INTERVAL_MINUTES = 30\n    STALE_EXPERIMENT_HOURS = 12.0\n    \n    MIN_MEMORY_GB = 2.0\n    MEMORY_CHECK_INTERVAL = 10\n    \n    MIN_ADDITIONAL_FEATURES = 1\n    MAX_ADDITIONAL_FEATURES = 30\n    CONFIGS_PER_FEATURE_SET = 50\n    \n    USE_FULL_DATASET = True\n    SAMPLE_SIZE_FOR_TESTING = None\n    \n    # Updated batch size as requested\n    BATCH_SIZE = 1024 * 8 * 4  # 32768\n    MAX_EPOCHS = 50\n    EARLY_STOPPING_PATIENCE = 7\n    \n    N_FOLDS = 7\n    MIN_FOLDS_FOR_VALID_SCORE = 5\n    \n    MAX_TRAIN_VAL_GAP = 0.05\n    MIN_ACCEPTABLE_VAL_SCORE = 0.001\n    MAX_ACCEPTABLE_TRAIN_SCORE = 0.35\n    MIN_VAL_STD = 0.0001\n    MAX_VAL_STD = 0.04\n    CONSISTENCY_THRESHOLD = 0.025\n    \n    MC_DROPOUT_SAMPLES = 5\n    MAX_PREDICTION_UNCERTAINTY = 0.15\n    \n    INFINITY_STRATEGIES = ['median', 'percentile', 'zero', 'winsorize']\n    \n    # Removed 'power' transform due to errors\n    FEATURE_TRANSFORMS = [\n        ['standard'],\n        ['rank'],\n        ['quantile'],\n        ['robust'],\n        ['standard', 'rank'],\n        ['robust', 'rank']\n    ]\n    \n    LABEL_TRANSFORMS = ['none', 'rank', 'quantile', 'log1p']\n    \n    # Complex architecture variants as requested\n    ARCHITECTURE_VARIANTS = [\n        {'name': 'baseline_mlp', 'hidden_dims': [256, 128, 64, 1], 'type': 'standard'},\n        {'name': 'deep', 'hidden_dims': [512, 256, 128, 64, 32, 1], 'type': 'deep'},\n        {'name': 'wide', 'hidden_dims': [1024, 256, 64, 1], 'type': 'wide'},\n        {'name': 'pyramid', 'hidden_dims': [256, 128, 64, 32, 16, 1], 'type': 'pyramid'},\n        {'name': 'bottleneck', 'hidden_dims': [512, 64, 512, 1], 'type': 'bottleneck'},\n        {'name': 'shallow_wide', 'hidden_dims': [2048, 512, 1], 'type': 'shallow'},\n        {'name': 'very_deep', 'hidden_dims': [256, 192, 128, 96, 64, 48, 32, 16, 1], 'type': 'very_deep'},\n        {'name': 'funnel', 'hidden_dims': [1024, 256, 64, 16, 1], 'type': 'funnel'},\n        {'name': 'ultra_deep', 'hidden_dims': [256, 224, 192, 160, 128, 96, 64, 48, 32, 16, 1], 'type': 'ultra_deep'},\n        {'name': 'expanding', 'hidden_dims': [64, 128, 256, 512, 256, 128, 64, 1], 'type': 'expanding'}\n    ]\n    \n    ACTIVATIONS = ['relu', 'tanh', 'leaky_relu', 'elu', 'selu', 'gelu']\n    \n    # Updated dropout rates closer to 0.5\n    DROPOUT_RATES = [0.4, 0.45, 0.5, 0.55, 0.6]\n    \n    # Updated learning rates closer to 0.001\n    LEARNING_RATES = [0.0005, 0.0008, 0.001, 0.0012, 0.0015]\n    \n    # Updated weight decays\n    WEIGHT_DECAYS = [0.0001, 0.0005, 0.001, 0.002, 0.005]\n    \n    # Updated input noise levels\n    INPUT_NOISE_LEVELS = [0.001, 0.003, 0.005, 0.008, 0.01]\n    \n    CV_STRATEGIES = ['kfold', 'stratified', 'random_split']\n    \n    USE_ENSEMBLE_UNCERTAINTY = True\n    ENSEMBLE_THRESHOLD = 0.05\n    \n    MAX_CONFIGS_TO_TRY = 2000\n\n@dataclass\nclass ExperimentConfig:\n    feature_list: List[str]\n    additional_features: List[str]\n    \n    infinity_strategy: str = 'median'\n    feature_transforms: List[str] = field(default_factory=lambda: ['standard'])\n    use_interactions: bool = False\n    interaction_features: List[str] = field(default_factory=list)\n    interaction_method: str = 'multiply'\n    use_clustering: bool = False\n    n_clusters: int = 10\n    use_pca: bool = False\n    pca_components: int = 20\n    use_binning: bool = False\n    binning_method: str = 'quantile'\n    binning_features: List[str] = field(default_factory=list)\n    n_bins: int = 10\n    \n    label_transform: str = 'none'\n    label_noise: float = 0.0\n    \n    architecture_name: str = 'baseline_mlp'\n    architecture_type: str = 'standard'\n    hidden_dims: List[int] = field(default_factory=lambda: [256, 128, 64, 1])\n    activation: str = 'relu'\n    dropout_rate: float = 0.5\n    dropout_rates_dict: Dict[str, float] = field(default_factory=dict)\n    noise_levels_dict: Dict[str, float] = field(default_factory=dict)\n    lr_multipliers_dict: Dict[str, float] = field(default_factory=dict)\n    use_batch_norm: bool = False\n    use_layer_norm: bool = False\n    use_residual: bool = False\n    use_spectral_norm: bool = False\n    use_gradient_penalty: bool = False\n    gradient_penalty_weight: float = 0.1\n    use_attention: bool = False\n    attention_heads: int = 4\n    \n    optimizer: str = 'adam'\n    learning_rate: float = 0.001\n    weight_decay: float = 0.001\n    use_scheduler: str = 'plateau'\n    \n    input_noise: float = 0.005\n    use_mixup: bool = False\n    mixup_alpha: float = 0.2\n    use_cutmix: bool = False\n    cutmix_alpha: float = 1.0\n    gradient_clip: float = 1.0\n    label_smoothing: float = 0.0\n    \n    cv_strategy: str = 'kfold'\n    n_ensemble: int = 1\n    use_mc_dropout: bool = True\n    \n    use_noise_detection: bool = False\n    noise_detection_method: str = 'ensemble'\n    \n    def get_hash(self) -> str:\n        config_dict = {\n            'features': sorted(self.additional_features),\n            'infinity': self.infinity_strategy,\n            'transforms': sorted(self.feature_transforms),\n            'interactions': self.use_interactions,\n            'interaction_features': sorted(self.interaction_features) if self.interaction_features else [],\n            'interaction_method': self.interaction_method,\n            'clustering': self.use_clustering,\n            'n_clusters': self.n_clusters,\n            'pca': self.use_pca,\n            'pca_components': self.pca_components,\n            'binning': self.use_binning,\n            'binning_method': self.binning_method,\n            'binning_features': sorted(self.binning_features) if self.binning_features else [],\n            'label_transform': self.label_transform,\n            'label_noise': self.label_noise,\n            'architecture': self.architecture_name,\n            'hidden_dims': self.hidden_dims,\n            'activation': self.activation,\n            'dropout': self.dropout_rate,\n            'batch_norm': self.use_batch_norm,\n            'layer_norm': self.use_layer_norm,\n            'residual': self.use_residual,\n            'spectral_norm': self.use_spectral_norm,\n            'gradient_penalty': self.use_gradient_penalty,\n            'attention': self.use_attention,\n            'optimizer': self.optimizer,\n            'lr': self.learning_rate,\n            'weight_decay': self.weight_decay,\n            'scheduler': self.use_scheduler,\n            'input_noise': self.input_noise,\n            'mixup': self.use_mixup,\n            'cutmix': self.use_cutmix,\n            'cv_strategy': self.cv_strategy,\n            'n_ensemble': self.n_ensemble,\n            'mc_dropout': self.use_mc_dropout,\n            'noise_detection': self.use_noise_detection\n        }\n        config_str = json.dumps(config_dict, sort_keys=True)\n        return hashlib.sha256(config_str.encode()).hexdigest()[:16]\n    \n    def get_description(self) -> str:\n        parts = []\n        \n        feature_group_config = Config.FEATURE_GROUP_CONFIG\n        \n        n_market = len([f for f in self.feature_list if f in feature_group_config.market_features])\n        n_micro = len([f for f in self.feature_list if f in feature_group_config.microstructure_features])\n        n_core_prop = len([f for f in self.feature_list if f in feature_group_config.core_proprietary_features])\n        n_other_prop = len([f for f in self.additional_features if f.startswith('X')])\n        \n        parts.append(f\"M:{n_market} Mi:{n_micro} CP:{n_core_prop} OP:{n_other_prop}\")\n        parts.append(f\"Total_Add:{len(self.additional_features)}\")\n        parts.append(f\"Inf:{self.infinity_strategy[:3]}\")\n        parts.append(f\"Arch:{self.architecture_name}\")\n        parts.append(f\"Drop:{self.dropout_rate}\")\n        parts.append(f\"LR:{self.learning_rate}\")\n        parts.append(f\"CV:{self.cv_strategy[:4]}\")\n        \n        if self.use_noise_detection:\n            parts.append(\"NoiseDetect\")\n        \n        return \" | \".join(parts)\n\ndef create_baseline_config() -> ExperimentConfig:\n    return ExperimentConfig(\n        feature_list=Config.CORE_FEATURES,\n        additional_features=[],\n        architecture_name='baseline_mlp',\n        architecture_type='standard',\n        hidden_dims=[256, 128, 64, 1],\n        activation='relu',\n        dropout_rate=0.5,\n        learning_rate=0.001,\n        weight_decay=0.001,\n        input_noise=0.005,\n        optimizer='adam',\n        use_batch_norm=False,\n        use_layer_norm=False,\n        use_spectral_norm=False,\n        use_gradient_penalty=False,\n        label_transform='none',\n        cv_strategy='kfold',\n        n_ensemble=1,\n        use_mc_dropout=True,\n        infinity_strategy='median',\n        feature_transforms=['standard'],\n        dropout_rates_dict=Config.FEATURE_GROUP_CONFIG.dropout_rates,\n        noise_levels_dict=Config.FEATURE_GROUP_CONFIG.noise_levels,\n        lr_multipliers_dict=Config.FEATURE_GROUP_CONFIG.lr_multipliers,\n        use_scheduler='plateau',\n        use_noise_detection=False\n    )\n\nclass MemoryManager:\n    def __init__(self):\n        self.gpu_available = torch.cuda.is_available()\n        self.device_count = torch.cuda.device_count() if self.gpu_available else 0\n        \n    def get_memory_info(self) -> Dict[str, float]:\n        info = {\n            'cpu_percent': psutil.virtual_memory().percent,\n            'cpu_available_gb': psutil.virtual_memory().available / 1e9,\n            'cpu_used_gb': psutil.virtual_memory().used / 1e9\n        }\n        \n        if self.gpu_available:\n            for i in range(self.device_count):\n                info[f'gpu_{i}_allocated_gb'] = torch.cuda.memory_allocated(i) / 1e9\n                info[f'gpu_{i}_reserved_gb'] = torch.cuda.memory_reserved(i) / 1e9\n                info[f'gpu_{i}_free_gb'] = (torch.cuda.get_device_properties(i).total_memory - \n                                           torch.cuda.memory_allocated(i)) / 1e9\n        \n        return info\n    \n    def clear_memory(self, device=None):\n        gc.collect()\n        if self.gpu_available:\n            if device is not None:\n                with torch.cuda.device(device):\n                    torch.cuda.empty_cache()\n                    torch.cuda.synchronize()\n            else:\n                for i in range(self.device_count):\n                    with torch.cuda.device(i):\n                        torch.cuda.empty_cache()\n                        torch.cuda.synchronize()\n    \n    def check_memory_available(self, required_gb: float = 2.0) -> bool:\n        info = self.get_memory_info()\n        return info['cpu_available_gb'] > required_gb\n    \n    def get_optimal_device(self) -> torch.device:\n        if not self.gpu_available:\n            return torch.device('cpu')\n        \n        return torch.device('cuda:0')\n\nclass NoiseDetector:\n    def __init__(self, config: NoiseDetectionConfig):\n        self.config = config\n        self.detectors = {}\n        \n    def fit(self, X: np.ndarray, y: np.ndarray):\n        print(\"Fitting noise detectors...\")\n        \n        # Ensure no NaN values\n        X_clean = np.nan_to_num(X, nan=0.0, posinf=0.0, neginf=0.0)\n        \n        if self.config.use_isolation_forest:\n            self.detectors['isolation_forest'] = IsolationForest(\n                contamination=self.config.isolation_contamination,\n                random_state=42,\n                n_jobs=-1\n            )\n            self.detectors['isolation_forest'].fit(X_clean)\n        \n        if self.config.use_local_outlier_factor:\n            # Fixed: Set novelty=True to enable predict()\n            self.detectors['lof'] = LocalOutlierFactor(\n                contamination=self.config.lof_contamination,\n                n_neighbors=min(20, X_clean.shape[0] - 1),\n                novelty=self.config.lof_novelty,\n                n_jobs=-1\n            )\n            self.detectors['lof'].fit(X_clean)\n        \n        if self.config.use_statistical_detection:\n            self.feature_means = np.mean(X_clean, axis=0)\n            self.feature_stds = np.std(X_clean, axis=0) + 1e-8\n        \n        if self.config.use_label_consistency:\n            self.label_mean = np.mean(y)\n            self.label_std = np.std(y) + 1e-8\n    \n    def predict(self, X: np.ndarray, y: Optional[np.ndarray] = None) -> np.ndarray:\n        n_samples = X.shape[0]\n        noise_votes = np.zeros(n_samples)\n        n_detectors = 0\n        \n        # Ensure no NaN values\n        X_clean = np.nan_to_num(X, nan=0.0, posinf=0.0, neginf=0.0)\n        \n        if 'isolation_forest' in self.detectors:\n            predictions = self.detectors['isolation_forest'].predict(X_clean)\n            noise_votes += (predictions == -1).astype(float)\n            n_detectors += 1\n        \n        if 'lof' in self.detectors:\n            # Fixed: Now we can use predict() since novelty=True\n            predictions = self.detectors['lof'].predict(X_clean)\n            noise_votes += (predictions == -1).astype(float)\n            n_detectors += 1\n        \n        if self.config.use_statistical_detection and hasattr(self, 'feature_means'):\n            z_scores = np.abs((X_clean - self.feature_means) / self.feature_stds)\n            max_z_scores = np.max(z_scores, axis=1)\n            noise_votes += (max_z_scores > self.config.statistical_threshold).astype(float)\n            n_detectors += 1\n        \n        if self.config.use_label_consistency and y is not None and hasattr(self, 'label_mean'):\n            label_z_scores = np.abs((y - self.label_mean) / self.label_std)\n            noise_votes += (label_z_scores > self.config.statistical_threshold).astype(float)\n            n_detectors += 1\n        \n        if n_detectors > 0:\n            noise_fraction = noise_votes / n_detectors\n            weights = 1 - noise_fraction\n            \n            weights[noise_fraction >= self.config.consensus_threshold] *= 0.5\n        else:\n            weights = np.ones(n_samples)\n        \n        return weights\n\nclass EnhancedSheetsTracker:\n    def __init__(self, credentials_file: str, spreadsheet_url: str, worker_id: str = None):\n        if not GSPREAD_AVAILABLE:\n            raise ImportError(\"gspread not available\")\n        \n        self.gc = gspread.service_account(filename=credentials_file)\n        self.spreadsheet = self.gc.open_by_url(spreadsheet_url)\n        \n        self.worker_id = worker_id or f\"{socket.gethostname()}_{os.getpid()}_{uuid.uuid4().hex[:8]}\"\n        \n        # Updated sheet name to V6\n        try:\n            self.sheet = self.spreadsheet.worksheet(\"Experiments_Regression_V6\")\n        except:\n            self.sheet = self.spreadsheet.add_worksheet(\"Experiments_Regression_V6\", 50000, 120)\n        \n        self.columns = [\n            'experiment_id', 'timestamp_start', 'timestamp_end', 'status',\n            'worker_id', 'config_hash', 'baseline_similarity_score',\n            \n            'core_features_list', 'n_core_features',\n            'market_features_list', 'n_market_features', \n            'microstructure_features_list', 'n_microstructure_features',\n            'additional_x_features_list', 'n_additional_x_features',\n            'total_features',\n            \n            'core_features_transform', 'market_features_transform',\n            'x_features_transform', 'global_transforms',\n            'infinity_strategy',\n            \n            'use_noise_detection', 'noise_detection_method', 'noise_samples_filtered',\n            \n            'use_interactions', 'interaction_pairs', 'n_interactions',\n            'interaction_method', 'interaction_features_source',\n            \n            'mean_train_score', 'mean_val_score', 'std_train_score', 'std_val_score',\n            'train_val_gap', 'best_fold_score', 'worst_fold_score',\n            'fold_consistency', 'n_valid_folds',\n            'improvement_over_baseline', 'overfitting_flag', 'overfitting_severity',\n            'mc_uncertainty', 'ensemble_uncertainty',\n            \n            'architecture_name', 'architecture_category', 'hidden_dims', \n            'total_parameters', 'depth', 'width',\n            \n            'activation', 'dropout_rates_dict', 'optimizer', 'learning_rate',\n            'weight_decay', 'noise_levels_dict', 'batch_size', 'input_noise',\n            'lr_multipliers_dict', 'use_spectral_norm', 'use_gradient_penalty',\n            'use_attention', 'attention_heads', 'label_smoothing',\n            \n            'use_clustering', 'n_clusters', 'use_pca', 'pca_components',\n            'use_binning', 'binning_details',\n            \n            'label_transform', 'cv_strategy', 'n_folds', 'n_ensemble',\n            'gradient_clip', 'use_mixup', 'use_cutmix',\n            'use_batch_norm', 'use_layer_norm', 'use_residual',\n            'use_mc_dropout', 'use_scheduler',\n            \n            'training_time_minutes', 'memory_peak_gb', 'gpu_memory_peak_gb',\n            'submission_file', 'error_message',\n            \n            'fold_scores_json', 'fold_train_scores_json', 'config_full_json'\n        ]\n        \n        self._ensure_headers()\n        \n        self.config_cache = {}\n        self._load_config_cache()\n        \n        self.experiment_rows = {}\n        \n        self.baseline_score = None\n        \n        all_values = self.sheet.get_all_values()\n        self.next_row = len(all_values) + 1\n        \n        print(f\"Initialized sheets tracker. Worker ID: {self.worker_id}\")\n        print(f\"Next available row: {self.next_row}\")\n        \n    def _ensure_headers(self):\n        try:\n            current_headers = self.sheet.row_values(1)\n            if current_headers != self.columns:\n                self.sheet.update('A1', [self.columns])\n                print(\"Updated spreadsheet headers\")\n        except:\n            self.sheet.update('A1', [self.columns])\n            print(\"Created spreadsheet headers\")\n    \n    def _load_config_cache(self):\n        try:\n            all_values = self.sheet.get_all_values()\n            if len(all_values) > 1:\n                header_row = all_values[0]\n                \n                hash_idx = header_row.index('config_hash')\n                status_idx = header_row.index('status')\n                timestamp_idx = header_row.index('timestamp_start')\n                worker_idx = header_row.index('worker_id') if 'worker_id' in header_row else None\n                \n                for row in all_values[1:]:\n                    if len(row) > hash_idx and row[hash_idx]:\n                        status = row[status_idx] if len(row) > status_idx else 'Unknown'\n                        timestamp = row[timestamp_idx] if len(row) > timestamp_idx else ''\n                        worker_id = row[worker_idx] if worker_idx and len(row) > worker_idx else 'Unknown'\n                        \n                        self.config_cache[row[hash_idx]] = {\n                            'status': status,\n                            'timestamp': timestamp,\n                            'worker_id': worker_id\n                        }\n            \n            print(f\"Loaded {len(self.config_cache)} existing configurations from cache\")\n            \n            self._mark_stale_experiments()\n            \n        except Exception as e:\n            print(f\"Warning: Failed to load config cache: {e}\")\n    \n    def _mark_stale_experiments(self):\n        now = datetime.now()\n        stale_count = 0\n        \n        for config_hash, info in self.config_cache.items():\n            if info['status'] in ['Started', 'Running']:\n                try:\n                    start_time = datetime.fromisoformat(info['timestamp'])\n                    if (now - start_time).total_seconds() / 3600 > Config.STALE_EXPERIMENT_HOURS:\n                        info['status'] = 'Did not complete in 12 hours'\n                        stale_count += 1\n                except:\n                    pass\n        \n        if stale_count > 0:\n            print(f\"Marked {stale_count} stale experiments\")\n    \n    def can_claim_config(self, config: ExperimentConfig) -> bool:\n        config_hash = config.get_hash()\n        \n        if config_hash not in self.config_cache:\n            return True\n        \n        info = self.config_cache[config_hash]\n        \n        if info['status'] in ['Did not complete in 12 hours', 'Error']:\n            return True\n        \n        if info['status'] in ['Completed', 'Started', 'Running']:\n            try:\n                start_time = datetime.fromisoformat(info['timestamp'])\n                if (datetime.now() - start_time).total_seconds() / 3600 > Config.STALE_EXPERIMENT_HOURS:\n                    return True\n            except:\n                pass\n            return False\n        \n        return True\n    \n    def atomic_claim_experiment(self, experiment_id: str, config: ExperimentConfig) -> Tuple[bool, int]:\n        config_hash = config.get_hash()\n        \n        self._load_config_cache()\n        \n        if not self.can_claim_config(config):\n            return False, -1\n        \n        row_data = self._prepare_row_data(config, {\n            'experiment_id': experiment_id,\n            'timestamp_start': datetime.now().isoformat(),\n            'timestamp_end': '',\n            'status': 'Started',\n            'config_hash': config_hash,\n            'worker_id': self.worker_id,\n            'baseline_similarity': 0\n        })\n        \n        row_list = []\n        for col in self.columns:\n            value = row_data.get(col, '')\n            row_list.append(str(value) if value is not None else '')\n        \n        try:\n            self.sheet.append_row(row_list, value_input_option='RAW')\n            row_number = self.next_row\n            self.next_row += 1\n            self.experiment_rows[experiment_id] = row_number\n            \n            self.config_cache[config_hash] = {\n                'status': 'Started',\n                'timestamp': datetime.now().isoformat(),\n                'worker_id': self.worker_id\n            }\n            \n            print(f\"✓ Claimed experiment {experiment_id} at row {row_number}\")\n            return True, row_number\n            \n        except Exception as e:\n            print(f\"✗ Failed to claim experiment: {e}\")\n            return False, -1\n    \n    def update_experiment_status(self, experiment_id: str, status: str, results: Dict[str, Any] = None):\n        if experiment_id not in self.experiment_rows:\n            print(f\"Warning: No row found for experiment {experiment_id}\")\n            return\n        \n        row_number = self.experiment_rows[experiment_id]\n        \n        try:\n            updates = []\n            \n            updates.append({\n                'range': f'D{row_number}',\n                'values': [[status]]\n            })\n            updates.append({\n                'range': f'C{row_number}',\n                'values': [[datetime.now().isoformat()]]\n            })\n            \n            if results:\n                col_indices = {col: i for i, col in enumerate(self.columns)}\n                \n                metrics = {\n                    'mean_val_score': results.get('mean_val_score', 0),\n                    'mean_train_score': results.get('mean_train_score', 0),\n                    'train_val_gap': results.get('train_val_gap', 0),\n                    'overfitting_flag': results.get('overfitting_flag', False),\n                    'overfitting_severity': results.get('overfitting_severity', 'None'),\n                    'training_time_minutes': results.get('training_time_minutes', 0),\n                    'fold_consistency': results.get('fold_consistency', 0),\n                    'n_valid_folds': results.get('n_valid_folds', 0),\n                    'mc_uncertainty': results.get('mc_uncertainty', 0),\n                    'ensemble_uncertainty': results.get('ensemble_uncertainty', 0),\n                    'noise_samples_filtered': results.get('noise_samples_filtered', 0)\n                }\n                \n                for metric, value in metrics.items():\n                    if metric in col_indices:\n                        col_letter = self._get_column_letter(col_indices[metric] + 1)\n                        updates.append({\n                            'range': f'{col_letter}{row_number}',\n                            'values': [[str(value)]]\n                        })\n            \n            self.sheet.batch_update(updates)\n            print(f\"✓ Updated experiment {experiment_id} status to {status}\")\n            \n        except Exception as e:\n            print(f\"✗ Failed to update experiment status: {e}\")\n            traceback.print_exc()\n    \n    def log_experiment_complete(self, experiment_id: str, config: ExperimentConfig, results: Dict[str, Any]):\n        if experiment_id not in self.experiment_rows:\n            print(f\"Warning: No row found for experiment {experiment_id}\")\n            return\n        \n        row_number = self.experiment_rows[experiment_id]\n        \n        row_data = self._prepare_row_data(config, results)\n        \n        row_list = []\n        for col in self.columns:\n            value = row_data.get(col, '')\n            row_list.append(str(value) if value is not None else '')\n        \n        try:\n            range_name = f'A{row_number}:{self._get_column_letter(len(self.columns))}{row_number}'\n            self.sheet.update(range_name, [row_list], value_input_option='RAW')\n            print(f\"✓ Updated complete results for experiment {experiment_id}\")\n        except Exception as e:\n            print(f\"✗ Failed to update experiment row: {e}\")\n            traceback.print_exc()\n    \n    def _prepare_row_data(self, config: ExperimentConfig, results: Dict[str, Any]) -> Dict[str, Any]:\n        config_hash = config.get_hash()\n        \n        feature_group_config = Config.FEATURE_GROUP_CONFIG\n        \n        market_features = [f for f in config.feature_list if f in feature_group_config.market_features]\n        microstructure_features = [f for f in config.feature_list if f in feature_group_config.microstructure_features]\n        core_proprietary = [f for f in config.feature_list if f in feature_group_config.core_proprietary_features]\n        additional_x = config.additional_features\n        \n        improvement = 0\n        if self.baseline_score and results.get('mean_val_score'):\n            improvement = results['mean_val_score'] - self.baseline_score\n        \n        total_params = 0\n        if config.hidden_dims:\n            prev_dim = len(config.feature_list)\n            for hidden_dim in config.hidden_dims:\n                total_params += prev_dim * hidden_dim + hidden_dim\n                prev_dim = hidden_dim\n        \n        fold_scores = to_python_type(results.get('fold_scores', []))\n        fold_train_scores = to_python_type(results.get('fold_train_scores', []))\n        \n        row_data = {\n            'experiment_id': results.get('experiment_id', ''),\n            'timestamp_start': results.get('timestamp_start', ''),\n            'timestamp_end': results.get('timestamp_end', datetime.now().isoformat()),\n            'status': results.get('status', 'completed'),\n            'worker_id': results.get('worker_id', self.worker_id),\n            'config_hash': config_hash,\n            'baseline_similarity_score': results.get('baseline_similarity', 0),\n            \n            'core_features_list': json.dumps(core_proprietary),\n            'n_core_features': len(core_proprietary),\n            'market_features_list': json.dumps(market_features),\n            'n_market_features': len(market_features),\n            'microstructure_features_list': json.dumps(microstructure_features),\n            'n_microstructure_features': len(microstructure_features),\n            'additional_x_features_list': json.dumps(additional_x),\n            'n_additional_x_features': len(additional_x),\n            'total_features': len(config.feature_list),\n            \n            'core_features_transform': 'standard',\n            'market_features_transform': 'standard',\n            'x_features_transform': json.dumps(config.feature_transforms),\n            'global_transforms': json.dumps(config.feature_transforms),\n            'infinity_strategy': config.infinity_strategy,\n            \n            'use_noise_detection': config.use_noise_detection,\n            'noise_detection_method': config.noise_detection_method,\n            'noise_samples_filtered': results.get('noise_samples_filtered', 0),\n            \n            'use_interactions': config.use_interactions,\n            'interaction_pairs': json.dumps(results.get('interaction_pairs', [])),\n            'n_interactions': len(results.get('interaction_pairs', [])),\n            'interaction_method': config.interaction_method,\n            'interaction_features_source': json.dumps(config.interaction_features),\n            \n            'mean_train_score': round(float(results.get('mean_train_score', 0)), 6),\n            'mean_val_score': round(float(results.get('mean_val_score', 0)), 6),\n            'std_train_score': round(float(results.get('std_train_score', 0)), 6),\n            'std_val_score': round(float(results.get('std_val_score', 0)), 6),\n            'train_val_gap': round(float(results.get('train_val_gap', 0)), 6),\n            'best_fold_score': round(float(results.get('best_fold_score', 0)), 6),\n            'worst_fold_score': round(float(results.get('worst_fold_score', 0)), 6),\n            'fold_consistency': round(float(results.get('fold_consistency', 0)), 6),\n            'n_valid_folds': results.get('n_valid_folds', 0),\n            'improvement_over_baseline': round(improvement, 6),\n            'overfitting_flag': results.get('overfitting_flag', False),\n            'overfitting_severity': results.get('overfitting_severity', 'None'),\n            'mc_uncertainty': round(float(results.get('mc_uncertainty', 0)), 6),\n            'ensemble_uncertainty': round(float(results.get('ensemble_uncertainty', 0)), 6),\n            \n            'architecture_name': config.architecture_name,\n            'architecture_category': config.architecture_type,\n            'hidden_dims': json.dumps(config.hidden_dims),\n            'total_parameters': total_params,\n            'depth': len(config.hidden_dims) if config.hidden_dims else 0,\n            'width': max(config.hidden_dims) if config.hidden_dims else 0,\n            \n            'activation': config.activation,\n            'dropout_rates_dict': json.dumps(config.dropout_rates_dict),\n            'optimizer': config.optimizer,\n            'learning_rate': config.learning_rate,\n            'weight_decay': config.weight_decay,\n            'noise_levels_dict': json.dumps(config.noise_levels_dict),\n            'batch_size': Config.BATCH_SIZE,\n            'input_noise': config.input_noise,\n            'lr_multipliers_dict': json.dumps(config.lr_multipliers_dict),\n            'use_spectral_norm': config.use_spectral_norm,\n            'use_gradient_penalty': config.use_gradient_penalty,\n            'use_attention': config.use_attention,\n            'attention_heads': config.attention_heads,\n            'label_smoothing': config.label_smoothing,\n            \n            'use_clustering': config.use_clustering,\n            'n_clusters': config.n_clusters,\n            'use_pca': config.use_pca,\n            'pca_components': config.pca_components,\n            'use_binning': config.use_binning,\n            'binning_details': json.dumps({\n                'method': config.binning_method,\n                'features': config.binning_features,\n                'n_bins': config.n_bins\n            }),\n            \n            'label_transform': config.label_transform,\n            'cv_strategy': config.cv_strategy,\n            'n_folds': results.get('n_folds', Config.N_FOLDS),\n            'n_ensemble': config.n_ensemble,\n            'gradient_clip': config.gradient_clip,\n            'use_mixup': config.use_mixup,\n            'use_cutmix': config.use_cutmix,\n            'use_batch_norm': config.use_batch_norm,\n            'use_layer_norm': config.use_layer_norm,\n            'use_residual': config.use_residual,\n            'use_mc_dropout': config.use_mc_dropout,\n            'use_scheduler': config.use_scheduler,\n            \n            'training_time_minutes': round(float(results.get('training_time_minutes', 0)), 2),\n            'memory_peak_gb': round(float(results.get('memory_peak_gb', 0)), 2),\n            'gpu_memory_peak_gb': round(float(results.get('gpu_memory_peak_gb', 0)), 2),\n            'submission_file': results.get('submission_file', ''),\n            'error_message': results.get('error_message', ''),\n            \n            'fold_scores_json': json.dumps(fold_scores),\n            'fold_train_scores_json': json.dumps(fold_train_scores),\n            'config_full_json': json.dumps(config.__dict__, default=str)\n        }\n        \n        return row_data\n    \n    def _get_column_letter(self, n):\n        string = \"\"\n        while n > 0:\n            n, remainder = divmod(n - 1, 26)\n            string = chr(65 + remainder) + string\n        return string\n    \n    def set_baseline_score(self, score: float):\n        self.baseline_score = score\n        print(f\"Set baseline score: {score:.6f}\")\n\n# Fixed SpectralNorm implementation\ndef spectral_norm(module, name='weight', n_power_iterations=1):\n    def _spectral_norm_forward_pre_hook(m, input):\n        weight = getattr(m, name)\n        if weight.ndim == 1:\n            return\n        \n        with torch.no_grad():\n            weight_mat = weight.view(weight.size(0), -1)\n            u = getattr(m, name + '_u')\n            v = getattr(m, name + '_v')\n            \n            for _ in range(n_power_iterations):\n                v = F.normalize(torch.mv(weight_mat.t(), u), dim=0)\n                u = F.normalize(torch.mv(weight_mat, v), dim=0)\n            \n            sigma = torch.dot(u, torch.mv(weight_mat, v))\n            weight.data = weight.data / sigma\n    \n    weight = getattr(module, name)\n    if weight.ndim == 1:\n        return module\n    \n    with torch.no_grad():\n        weight_mat = weight.view(weight.size(0), -1)\n        h, w = weight_mat.size()\n        u = F.normalize(weight.new_empty(h).normal_(0, 1), dim=0)\n        v = F.normalize(weight.new_empty(w).normal_(0, 1), dim=0)\n    \n    delattr(module, name)\n    module.register_parameter(name, nn.Parameter(weight.data))\n    module.register_buffer(name + '_u', u)\n    module.register_buffer(name + '_v', v)\n    module.register_forward_pre_hook(_spectral_norm_forward_pre_hook)\n    \n    return module\n\nclass SelfAttention(nn.Module):\n    def __init__(self, dim: int, heads: int = 4):\n        super().__init__()\n        self.heads = heads\n        self.head_dim = dim // heads\n        \n        # Ensure dim is divisible by heads\n        assert dim % heads == 0, f\"dim {dim} must be divisible by heads {heads}\"\n        \n        self.qkv = nn.Linear(dim, dim * 3, bias=False)\n        self.out = nn.Linear(dim, dim)\n        \n    def forward(self, x):\n        B, D = x.shape\n        \n        # Generate Q, K, V\n        qkv = self.qkv(x)  # [B, 3*D]\n        qkv = qkv.reshape(B, 3, self.heads, self.head_dim)  # [B, 3, heads, head_dim]\n        q, k, v = qkv.unbind(1)  # Each is [B, heads, head_dim]\n        \n        # Compute attention scores\n        attn = torch.einsum('bhd,bhd->bh', q, k) * (self.head_dim ** -0.5)  # [B, heads]\n        attn = attn.softmax(dim=-1)  # [B, heads]\n        \n        # Apply attention to values\n        out = torch.einsum('bh,bhd->bhd', attn, v)  # [B, heads, head_dim]\n        out = out.reshape(B, D)  # [B, D]\n        \n        return self.out(out)\n\nclass HierarchicalMLP(nn.Module):\n    def __init__(self, input_dim: int, config: ExperimentConfig, feature_indices: Dict[str, List[int]]):\n        super().__init__()\n        \n        self.config = config\n        self.feature_indices = feature_indices\n        \n        self.group_processors = nn.ModuleDict()\n        \n        # Process each feature group\n        for group_name, indices in feature_indices.items():\n            if len(indices) > 0:\n                group_dim = len(indices)\n                hidden_dim = min(group_dim * 2, 64)\n                \n                layers = [\n                    nn.Linear(group_dim, hidden_dim),\n                    self._get_activation(config.activation),\n                    nn.Dropout(config.dropout_rates_dict.get(group_name, 0.5)),\n                    nn.Linear(hidden_dim, hidden_dim // 2)\n                ]\n                \n                self.group_processors[group_name] = nn.Sequential(*layers)\n        \n        # Calculate processed dimension\n        processed_dim = sum(\n            min(len(indices) * 2, 64) // 2 \n            for indices in feature_indices.values() \n            if len(indices) > 0\n        )\n        \n        # Attention layer (if enabled)\n        if config.use_attention and processed_dim > 0:\n            # Ensure processed_dim is divisible by attention_heads\n            if processed_dim % config.attention_heads != 0:\n                # Adjust processed_dim to be divisible\n                processed_dim = ((processed_dim // config.attention_heads) + 1) * config.attention_heads\n                self.projection = nn.Linear(\n                    sum(min(len(indices) * 2, 64) // 2 \n                        for indices in feature_indices.values() \n                        if len(indices) > 0),\n                    processed_dim\n                )\n            else:\n                self.projection = None\n            \n            self.attention = SelfAttention(processed_dim, config.attention_heads)\n            self.attention_dropout = nn.Dropout(config.dropout_rate * 0.5)\n        else:\n            self.attention = None\n            self.projection = None\n        \n        # Main network\n        self.main_network = self._build_main_network(processed_dim, config)\n        \n        self.noise_levels = config.noise_levels_dict\n        \n    def _get_activation(self, activation: str):\n        if activation == 'relu':\n            return nn.ReLU()\n        elif activation == 'tanh':\n            return nn.Tanh()\n        elif activation == 'leaky_relu':\n            return nn.LeakyReLU(0.1)\n        elif activation == 'elu':\n            return nn.ELU()\n        elif activation == 'selu':\n            return nn.SELU()\n        elif activation == 'gelu':\n            return nn.GELU()\n        else:\n            return nn.ReLU()\n    \n    def _build_main_network(self, input_dim: int, config: ExperimentConfig):\n        layers = []\n        prev_dim = input_dim\n        \n        for i, hidden_dim in enumerate(config.hidden_dims[:-1]):\n            linear = nn.Linear(prev_dim, hidden_dim)\n            \n            if config.use_spectral_norm:\n                linear = spectral_norm(linear)\n            \n            layers.append(linear)\n            \n            if config.use_batch_norm:\n                layers.append(nn.BatchNorm1d(hidden_dim))\n            elif config.use_layer_norm:\n                layers.append(nn.LayerNorm(hidden_dim))\n            \n            layers.append(self._get_activation(config.activation))\n            \n            dropout_rate = min(config.dropout_rate + i * 0.05, 0.9)\n            layers.append(nn.Dropout(dropout_rate))\n            \n            if config.use_residual and prev_dim == hidden_dim:\n                block = nn.Sequential(*layers[-4:])\n                layers = layers[:-4]\n                layers.append(ResidualBlock(block))\n            \n            prev_dim = hidden_dim\n        \n        layers.append(nn.Linear(prev_dim, config.hidden_dims[-1]))\n        \n        return nn.Sequential(*layers)\n    \n    def forward(self, x):\n        group_outputs = []\n        \n        # Process each feature group\n        for group_name, indices in self.feature_indices.items():\n            if len(indices) > 0:\n                group_features = x[:, indices]\n                \n                # Add noise during training\n                if self.training:\n                    noise_level = self.noise_levels.get(group_name, 0.005)\n                    noise = torch.randn_like(group_features) * noise_level\n                    group_features = group_features + noise\n                \n                group_output = self.group_processors[group_name](group_features)\n                group_outputs.append(group_output)\n        \n        if group_outputs:\n            combined = torch.cat(group_outputs, dim=1)\n            \n            # Apply projection if needed for attention\n            if self.projection is not None:\n                combined = self.projection(combined)\n            \n            # Apply attention if enabled\n            if self.attention is not None:\n                attended = self.attention(combined)\n                combined = combined + self.attention_dropout(attended)\n        else:\n            combined = x\n        \n        # Pass through main network\n        output = self.main_network(combined)\n        \n        return output.squeeze()\n\nclass ResidualBlock(nn.Module):\n    def __init__(self, block):\n        super().__init__()\n        self.block = block\n    \n    def forward(self, x):\n        return x + self.block(x)\n\nclass CryptoModel(nn.Module):\n    def __init__(self, input_dim: int, config: ExperimentConfig, feature_indices: Dict[str, List[int]] = None):\n        super().__init__()\n        self.config = config\n        self.input_noise = config.input_noise\n        self.use_mc_dropout = config.use_mc_dropout\n        \n        if feature_indices and any(len(idx) > 0 for idx in feature_indices.values()):\n            self.model = HierarchicalMLP(input_dim, config, feature_indices)\n        else:\n            self.model = self._build_standard_model(input_dim, config)\n        \n        self.apply(self._init_weights)\n    \n    def _get_activation(self, activation: str):\n        if activation == 'relu':\n            return nn.ReLU()\n        elif activation == 'tanh':\n            return nn.Tanh()\n        elif activation == 'leaky_relu':\n            return nn.LeakyReLU(0.1)\n        elif activation == 'elu':\n            return nn.ELU()\n        elif activation == 'selu':\n            return nn.SELU()\n        elif activation == 'gelu':\n            return nn.GELU()\n        else:\n            return nn.ReLU()\n    \n    def _build_standard_model(self, input_dim: int, config: ExperimentConfig):\n        layers = []\n        prev_dim = input_dim\n        \n        for i, hidden_dim in enumerate(config.hidden_dims[:-1]):\n            linear = nn.Linear(prev_dim, hidden_dim)\n            \n            if config.use_spectral_norm:\n                linear = spectral_norm(linear)\n            \n            layers.append(linear)\n            \n            if config.use_batch_norm:\n                layers.append(nn.BatchNorm1d(hidden_dim))\n            elif config.use_layer_norm:\n                layers.append(nn.LayerNorm(hidden_dim))\n            \n            layers.append(self._get_activation(config.activation))\n            \n            dropout_rate = min(config.dropout_rate + i * 0.05, 0.9)\n            layers.append(nn.Dropout(dropout_rate))\n            \n            prev_dim = hidden_dim\n        \n        layers.append(nn.Linear(prev_dim, config.hidden_dims[-1]))\n        \n        return nn.Sequential(*layers)\n    \n    def _init_weights(self, m):\n        if isinstance(m, nn.Linear):\n            if self.config.activation in ['relu', 'leaky_relu', 'elu', 'selu']:\n                nn.init.kaiming_normal_(m.weight, mode='fan_out', \n                                       nonlinearity='relu' if self.config.activation != 'selu' else 'linear')\n            else:\n                nn.init.xavier_normal_(m.weight, gain=0.5)\n            \n            if m.bias is not None:\n                nn.init.constant_(m.bias, 0)\n    \n    def forward(self, x, enable_dropout=False):\n        if self.training and self.input_noise > 0:\n            noise = torch.randn_like(x) * self.input_noise\n            x = x + noise\n        \n        if enable_dropout and self.use_mc_dropout:\n            self.train()\n            output = self.model(x)\n            self.eval()\n            return output\n        else:\n            return self.model(x)\n    \n    def mc_forward(self, x, n_samples=10):\n        if not self.use_mc_dropout:\n            return self.forward(x), torch.zeros(x.shape[0], device=x.device)\n        \n        outputs = []\n        for _ in range(n_samples):\n            outputs.append(self.forward(x, enable_dropout=True))\n        \n        outputs = torch.stack(outputs)\n        mean = outputs.mean(dim=0)\n        uncertainty = outputs.std(dim=0)\n        \n        return mean, uncertainty\n\nclass FeatureEngineer:\n    def __init__(self, config: ExperimentConfig):\n        self.config = config\n        self.fitted = False\n        self.transformers = {}\n        self.feature_stats = {}\n        \n    def fit(self, X: pd.DataFrame, y: np.ndarray):\n        for col in X.columns:\n            values = X[col].values\n            finite_mask = np.isfinite(values)\n            if finite_mask.sum() > 0:\n                finite_values = values[finite_mask]\n                self.feature_stats[col] = {\n                    'median': np.median(finite_values),\n                    'mean': np.mean(finite_values),\n                    'std': np.std(finite_values),\n                    'p5': np.percentile(finite_values, 5),\n                    'p95': np.percentile(finite_values, 95),\n                    'p25': np.percentile(finite_values, 25),\n                    'p75': np.percentile(finite_values, 75),\n                    'p1': np.percentile(finite_values, 1),\n                    'p99': np.percentile(finite_values, 99)\n                }\n        \n        X_clean = self._handle_infinity(X.copy())\n        \n        if 'standard' in self.config.feature_transforms:\n            self.transformers['standard'] = StandardScaler()\n            self.transformers['standard'].fit(X_clean)\n        \n        if 'robust' in self.config.feature_transforms:\n            self.transformers['robust'] = RobustScaler()\n            self.transformers['robust'].fit(X_clean)\n        \n        if 'quantile' in self.config.feature_transforms:\n            self.transformers['quantile'] = QuantileTransformer(\n                n_quantiles=min(1000, len(X_clean)),\n                output_distribution='normal'\n            )\n            self.transformers['quantile'].fit(X_clean)\n        \n        # Removed PowerTransformer due to errors\n        \n        self.fitted = True\n    \n    def transform(self, X: pd.DataFrame) -> np.ndarray:\n        if not self.fitted:\n            raise ValueError(\"Must fit before transform\")\n        \n        X_clean = self._handle_infinity(X.copy())\n        \n        all_features = []\n        \n        if 'standard' in self.config.feature_transforms:\n            all_features.append(self.transformers['standard'].transform(X_clean))\n        \n        if 'robust' in self.config.feature_transforms and 'robust' in self.transformers:\n            all_features.append(self.transformers['robust'].transform(X_clean))\n        \n        if 'rank' in self.config.feature_transforms:\n            rank_features = np.apply_along_axis(\n                lambda x: rankdata(x, method='average') / (len(x) + 1), 0, X_clean.values\n            )\n            all_features.append(rank_features)\n        \n        if 'quantile' in self.config.feature_transforms and 'quantile' in self.transformers:\n            all_features.append(self.transformers['quantile'].transform(X_clean))\n        \n        if len(all_features) == 0:\n            return X_clean.values\n        \n        return np.hstack(all_features)\n    \n    def _handle_infinity(self, X: pd.DataFrame) -> pd.DataFrame:\n        for col in X.columns:\n            if col not in self.feature_stats:\n                finite_mask = np.isfinite(X[col])\n                if finite_mask.sum() > 0:\n                    median_val = np.median(X[col][finite_mask])\n                    X.loc[~finite_mask, col] = median_val\n                else:\n                    X[col] = 0\n                continue\n            \n            stats = self.feature_stats[col]\n            \n            if self.config.infinity_strategy == 'median':\n                X.loc[~np.isfinite(X[col]), col] = stats['median']\n            elif self.config.infinity_strategy == 'percentile':\n                X.loc[X[col] == np.inf, col] = stats['p95']\n                X.loc[X[col] == -np.inf, col] = stats['p5']\n                X.loc[np.isnan(X[col]), col] = stats['median']\n            elif self.config.infinity_strategy == 'zero':\n                X.loc[~np.isfinite(X[col]), col] = 0\n            elif self.config.infinity_strategy == 'winsorize':\n                X.loc[X[col] > stats['p99'], col] = stats['p99']\n                X.loc[X[col] < stats['p1'], col] = stats['p1']\n                X.loc[np.isnan(X[col]), col] = stats['median']\n        \n        return X\n\nclass ModelTrainer:\n    def __init__(self, config: ExperimentConfig, memory_manager: MemoryManager):\n        self.config = config\n        self.memory_manager = memory_manager\n        \n    def train_with_cv(self, X_train: np.ndarray, y_train: np.ndarray, \n                      feature_indices: Dict[str, List[int]] = None,\n                      sample_weights: Optional[np.ndarray] = None) -> Dict[str, Any]:\n        \n        fold_train_scores = []\n        fold_val_scores = []\n        models = []\n        mc_uncertainties = []\n        \n        noise_detector = None\n        noise_samples_filtered = 0\n        \n        if self.config.use_noise_detection:\n            noise_config = Config.NOISE_DETECTION_CONFIG\n            noise_detector = NoiseDetector(noise_config)\n            noise_detector.fit(X_train, y_train)\n            \n            if sample_weights is None:\n                sample_weights = noise_detector.predict(X_train, y_train)\n            else:\n                sample_weights *= noise_detector.predict(X_train, y_train)\n            \n            noise_samples_filtered = np.sum(sample_weights < 0.5)\n            print(f\"Identified {noise_samples_filtered} potentially noisy samples\")\n        \n        # Choose CV strategy\n        if self.config.cv_strategy == 'kfold':\n            cv = KFold(n_splits=Config.N_FOLDS, shuffle=True, random_state=42)\n            splits = list(cv.split(X_train))\n            \n        elif self.config.cv_strategy == 'stratified':\n            y_bins = pd.qcut(y_train, q=10, labels=False, duplicates='drop')\n            cv = StratifiedKFold(n_splits=Config.N_FOLDS, shuffle=True, random_state=42)\n            splits = list(cv.split(X_train, y_bins))\n            \n        elif self.config.cv_strategy == 'random_split':\n            n_samples = len(X_train)\n            indices = np.arange(n_samples)\n            np.random.shuffle(indices)\n            splits = []\n            \n            fold_size = n_samples // Config.N_FOLDS\n            for i in range(Config.N_FOLDS):\n                val_start = i * fold_size\n                val_end = (i + 1) * fold_size if i < Config.N_FOLDS - 1 else n_samples\n                val_idx = indices[val_start:val_end]\n                train_idx = np.concatenate([indices[:val_start], indices[val_end:]])\n                splits.append((train_idx, val_idx))\n        \n        print(f\"\\nTraining with {len(splits)} folds using {self.config.cv_strategy} strategy\")\n        \n        valid_folds = 0\n        \n        for fold_idx, (train_idx, val_idx) in enumerate(splits):\n            print(f\"\\nFold {fold_idx + 1}/{len(splits)}\")\n            \n            if fold_idx > 0 and fold_idx % Config.CLEAR_MEMORY_INTERVAL == 0:\n                self.memory_manager.clear_memory()\n                print(\"  Cleared GPU memory\")\n            \n            X_fold_train = X_train[train_idx]\n            y_fold_train = y_train[train_idx]\n            X_fold_val = X_train[val_idx]\n            y_fold_val = y_train[val_idx]\n            \n            fold_sample_weights = None\n            if sample_weights is not None:\n                fold_sample_weights = sample_weights[train_idx]\n            \n            try:\n                result = self._train_single_model(\n                    X_fold_train, y_fold_train, X_fold_val, y_fold_val,\n                    feature_indices=feature_indices,\n                    sample_weights=fold_sample_weights,\n                    seed=42 + fold_idx\n                )\n                \n                train_score, val_score, model, mc_uncertainty = result\n                \n                if val_score > Config.MIN_ACCEPTABLE_VAL_SCORE and not np.isnan(val_score):\n                    valid_folds += 1\n                    fold_train_scores.append(train_score)\n                    fold_val_scores.append(val_score)\n                    models.append(model)\n                    mc_uncertainties.append(mc_uncertainty)\n                    \n                    print(f\"  Train: {train_score:.6f}, Val: {val_score:.6f}, Gap: {train_score - val_score:.6f}\")\n                    print(f\"  MC Uncertainty: {mc_uncertainty:.6f}\")\n                else:\n                    print(f\"  Fold invalid - Val score: {val_score:.6f}\")\n                \n            except Exception as e:\n                print(f\"  Fold failed: {str(e)}\")\n                continue\n            \n            if fold_idx + 1 - valid_folds > Config.N_FOLDS - Config.MIN_FOLDS_FOR_VALID_SCORE:\n                print(\"  Too many invalid folds, stopping CV\")\n                break\n        \n        if not fold_val_scores:\n            print(\"Warning: No valid folds found\")\n            fold_val_scores = [0.0]\n            fold_train_scores = [0.0]\n        \n        fold_consistency = 0\n        if len(fold_val_scores) > 1:\n            fold_consistency = np.max(fold_val_scores) - np.min(fold_val_scores)\n        \n        results = {\n            'fold_train_scores': fold_train_scores,\n            'fold_scores': fold_val_scores,\n            'mean_train_score': np.mean(fold_train_scores) if fold_train_scores else 0,\n            'mean_val_score': np.mean(fold_val_scores) if fold_val_scores else 0,\n            'std_train_score': np.std(fold_train_scores) if len(fold_train_scores) > 1 else 0,\n            'std_val_score': np.std(fold_val_scores) if len(fold_val_scores) > 1 else 0,\n            'train_val_gap': (np.mean(fold_train_scores) - np.mean(fold_val_scores)) if fold_train_scores else 0,\n            'best_fold_score': np.max(fold_val_scores) if fold_val_scores else 0,\n            'worst_fold_score': np.min(fold_val_scores) if fold_val_scores else 0,\n            'fold_consistency': fold_consistency,\n            'n_valid_folds': valid_folds,\n            'models': models,\n            'n_folds': len(fold_val_scores),\n            'mc_uncertainty': np.mean(mc_uncertainties) if mc_uncertainties else 0,\n            'overfitting_flag': False,\n            'overfitting_severity': 'None',\n            'noise_samples_filtered': noise_samples_filtered\n        }\n        \n        # Check for overfitting\n        if results['mean_train_score'] > 0 and results['mean_val_score'] > 0:\n            if results['train_val_gap'] > Config.MAX_TRAIN_VAL_GAP:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'High'\n            \n            if results['mean_train_score'] > Config.MAX_ACCEPTABLE_TRAIN_SCORE:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'Very High'\n            \n            if results['std_val_score'] > Config.MAX_VAL_STD:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'Unstable'\n            \n            if results['fold_consistency'] > Config.CONSISTENCY_THRESHOLD:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'Inconsistent'\n            \n            if results['n_valid_folds'] < Config.MIN_FOLDS_FOR_VALID_SCORE:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'Insufficient Valid Folds'\n            \n            if results['mean_train_score'] > 2 * results['mean_val_score'] and results['mean_val_score'] > 0:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'Severe'\n            \n            if results['mc_uncertainty'] > Config.MAX_PREDICTION_UNCERTAINTY:\n                results['overfitting_flag'] = True\n                results['overfitting_severity'] = 'High Uncertainty'\n        \n        return results\n    \n    def _train_single_model(self, X_train: np.ndarray, y_train: np.ndarray,\n                           X_val: np.ndarray, y_val: np.ndarray,\n                           feature_indices: Dict[str, List[int]] = None,\n                           sample_weights: Optional[np.ndarray] = None,\n                           seed: int = 42) -> Tuple[float, float, nn.Module, float]:\n        torch.manual_seed(seed)\n        np.random.seed(seed)\n        \n        device = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n        \n        # Transform labels\n        y_train_transformed = self._transform_labels(y_train)\n        y_val_transformed = self._transform_labels(y_val)\n        \n        # Create model\n        model = CryptoModel(X_train.shape[1], self.config, feature_indices).to(device)\n        \n        # Set up parameter groups\n        param_groups = []\n        \n        base_model = model\n        \n        if hasattr(base_model.model, 'group_processors'):\n            for group_name, processor in base_model.model.group_processors.items():\n                lr_mult = self.config.lr_multipliers_dict.get(group_name, 1.0)\n                param_groups.append({\n                    'params': processor.parameters(),\n                    'lr': self.config.learning_rate * lr_mult,\n                    'weight_decay': self.config.weight_decay\n                })\n            \n            param_groups.append({\n                'params': base_model.model.main_network.parameters(),\n                'lr': self.config.learning_rate,\n                'weight_decay': self.config.weight_decay\n            })\n        else:\n            param_groups.append({\n                'params': model.parameters(),\n                'lr': self.config.learning_rate,\n                'weight_decay': self.config.weight_decay\n            })\n        \n        # Create optimizer\n        if self.config.optimizer == 'adam':\n            optimizer = optim.Adam(param_groups)\n        elif self.config.optimizer == 'adamw':\n            optimizer = optim.AdamW(param_groups)\n        elif self.config.optimizer == 'sgd':\n            optimizer = optim.SGD(param_groups, momentum=0.9)\n        else:\n            optimizer = optim.Adam(param_groups)\n        \n        # Create scheduler\n        if self.config.use_scheduler == 'plateau':\n            scheduler = optim.lr_scheduler.ReduceLROnPlateau(\n                optimizer, mode='max', patience=3, factor=0.5, min_lr=1e-7\n            )\n        elif self.config.use_scheduler == 'cosine':\n            scheduler = optim.lr_scheduler.CosineAnnealingLR(\n                optimizer, T_max=Config.MAX_EPOCHS, eta_min=1e-7\n            )\n        elif self.config.use_scheduler == 'exponential':\n            scheduler = optim.lr_scheduler.ExponentialLR(optimizer, gamma=0.95)\n        else:\n            scheduler = None\n        \n        # Create datasets\n        train_dataset = TensorDataset(\n            torch.FloatTensor(X_train),\n            torch.FloatTensor(y_train_transformed)\n        )\n        \n        if sample_weights is not None:\n            train_dataset = TensorDataset(\n                torch.FloatTensor(X_train),\n                torch.FloatTensor(y_train_transformed),\n                torch.FloatTensor(sample_weights)\n            )\n        \n        val_dataset = TensorDataset(\n            torch.FloatTensor(X_val),\n            torch.FloatTensor(y_val_transformed)\n        )\n        \n        # Create data loaders with smaller batch size if memory issues\n        try:\n            train_loader = DataLoader(\n                train_dataset, \n                batch_size=Config.BATCH_SIZE, \n                shuffle=True,\n                num_workers=0,\n                pin_memory=True if device.type == 'cuda' else False\n            )\n        except:\n            # If batch size is too large, use smaller one\n            train_loader = DataLoader(\n                train_dataset, \n                batch_size=min(Config.BATCH_SIZE, 8192), \n                shuffle=True,\n                num_workers=0,\n                pin_memory=False\n            )\n        \n        val_loader = DataLoader(\n            val_dataset, \n            batch_size=min(Config.BATCH_SIZE * 2, 16384), \n            shuffle=False,\n            num_workers=0,\n            pin_memory=False\n        )\n        \n        # Loss function\n        criterion = nn.HuberLoss(delta=5.0)\n        \n        # Scaler for mixed precision\n        scaler = GradScaler() if Config.USE_MIXED_PRECISION and device.type == 'cuda' else None\n        \n        best_val_score = -np.inf\n        patience_counter = 0\n        best_model_state = None\n        \n        # Training loop\n        for epoch in range(Config.MAX_EPOCHS):\n            model.train()\n            train_losses = []\n            \n            for batch_data in train_loader:\n                if sample_weights is not None:\n                    batch_X, batch_y, batch_weights = batch_data\n                    batch_weights = batch_weights.to(device)\n                else:\n                    batch_X, batch_y = batch_data\n                    batch_weights = None\n                \n                batch_X, batch_y = batch_X.to(device), batch_y.to(device)\n                \n                optimizer.zero_grad()\n                \n                if scaler is not None:\n                    with autocast():\n                        outputs = model(batch_X)\n                        \n                        if self.config.label_smoothing > 0:\n                            smooth_targets = batch_y * (1 - self.config.label_smoothing) + \\\n                                           outputs.detach().mean() * self.config.label_smoothing\n                            loss = criterion(outputs, smooth_targets)\n                        else:\n                            loss = criterion(outputs, batch_y)\n                        \n                        if batch_weights is not None:\n                            loss = (loss * batch_weights).mean()\n                    \n                    scaler.scale(loss).backward()\n                    \n                    scaler.unscale_(optimizer)\n                    torch.nn.utils.clip_grad_norm_(model.parameters(), self.config.gradient_clip)\n                    \n                    scaler.step(optimizer)\n                    scaler.update()\n                else:\n                    outputs = model(batch_X)\n                    \n                    if self.config.label_smoothing > 0:\n                        smooth_targets = batch_y * (1 - self.config.label_smoothing) + \\\n                                       outputs.detach().mean() * self.config.label_smoothing\n                        loss = criterion(outputs, smooth_targets)\n                    else:\n                        loss = criterion(outputs, batch_y)\n                    \n                    if batch_weights is not None:\n                        loss = (loss * batch_weights).mean()\n                    \n                    loss.backward()\n                    \n                    torch.nn.utils.clip_grad_norm_(model.parameters(), self.config.gradient_clip)\n                    \n                    optimizer.step()\n                \n                train_losses.append(loss.item())\n            \n            # Validation\n            model.eval()\n            val_preds = []\n            val_targets = []\n            val_uncertainties = []\n            \n            with torch.no_grad():\n                for batch_X, batch_y in val_loader:\n                    batch_X = batch_X.to(device)\n                    \n                    if self.config.use_mc_dropout:\n                        preds, uncertainty = base_model.mc_forward(\n                            batch_X, n_samples=Config.MC_DROPOUT_SAMPLES\n                        )\n                        val_uncertainties.extend(uncertainty.detach().cpu().numpy())\n                    else:\n                        preds = model(batch_X)\n                    \n                    val_preds.extend(preds.detach().cpu().numpy())\n                    val_targets.extend(batch_y.numpy())\n            \n            val_preds = np.array(val_preds)\n            val_targets = np.array(val_targets)\n            mc_uncertainty = np.mean(val_uncertainties) if val_uncertainties else 0\n            \n            # Calculate validation score\n            if len(val_preds) > 1 and len(np.unique(val_preds)) > 1:\n                val_score = pearsonr(val_targets, val_preds)[0]\n                val_score = 0 if np.isnan(val_score) else val_score\n            else:\n                val_score = 0\n            \n            # Calculate train score on subset\n            model.eval()\n            with torch.no_grad():\n                train_subset_size = min(len(X_train), 10000)\n                train_subset_idx = np.random.choice(len(X_train), train_subset_size, replace=False)\n                train_subset = torch.FloatTensor(X_train[train_subset_idx]).to(device)\n                train_preds_subset = model(train_subset).detach().cpu().numpy()\n                \n                if len(train_preds_subset) > 1 and len(np.unique(train_preds_subset)) > 1:\n                    train_score = pearsonr(\n                        y_train_transformed[train_subset_idx], \n                        train_preds_subset\n                    )[0]\n                    train_score = 0 if np.isnan(train_score) else train_score\n                else:\n                    train_score = 0\n            \n            # Early stopping checks\n            if train_score - val_score > Config.MAX_TRAIN_VAL_GAP * 2:\n                print(f\"    Epoch {epoch}: Stopping due to overfitting\")\n                break\n            \n            if train_score > Config.MAX_ACCEPTABLE_TRAIN_SCORE:\n                print(f\"    Epoch {epoch}: Train score too high ({train_score:.4f})\")\n                break\n            \n            # Update learning rate\n            if scheduler is not None:\n                if self.config.use_scheduler == 'plateau':\n                    scheduler.step(val_score)\n                else:\n                    scheduler.step()\n            \n            # Save best model\n            if val_score > best_val_score:\n                best_val_score = val_score\n                patience_counter = 0\n                best_model_state = model.state_dict().copy()\n            else:\n                patience_counter += 1\n                if patience_counter >= Config.EARLY_STOPPING_PATIENCE:\n                    break\n        \n        # Load best model\n        if best_model_state is not None:\n            model.load_state_dict(best_model_state)\n        \n        # Final evaluation\n        model.eval()\n        with torch.no_grad():\n            final_train_preds = []\n            for i in range(0, len(X_train), Config.MAX_BATCH_SIZE_PREDICT):\n                batch = torch.FloatTensor(\n                    X_train[i:i+Config.MAX_BATCH_SIZE_PREDICT]\n                ).to(device)\n                final_train_preds.extend(model(batch).detach().cpu().numpy())\n            \n            if len(final_train_preds) > 1 and len(np.unique(final_train_preds)) > 1:\n                train_score = pearsonr(y_train_transformed, final_train_preds)[0]\n                train_score = 0 if np.isnan(train_score) else train_score\n            else:\n                train_score = 0\n            \n            # Calculate MC uncertainty\n            if self.config.use_mc_dropout:\n                val_uncertainties = []\n                for i in range(0, len(X_val), Config.MAX_BATCH_SIZE_PREDICT):\n                    batch = torch.FloatTensor(\n                        X_val[i:i+Config.MAX_BATCH_SIZE_PREDICT]\n                    ).to(device)\n                    _, uncertainty = base_model.mc_forward(\n                        batch, n_samples=Config.MC_DROPOUT_SAMPLES\n                    )\n                    val_uncertainties.extend(uncertainty.detach().cpu().numpy())\n                mc_uncertainty = np.mean(val_uncertainties)\n            else:\n                mc_uncertainty = 0\n        \n        model = model.cpu()\n        \n        return train_score, best_val_score, model, mc_uncertainty\n    \n    def _transform_labels(self, y: np.ndarray) -> np.ndarray:\n        if self.config.label_transform == 'none':\n            return y\n        elif self.config.label_transform == 'rank':\n            return rankdata(y) / (len(y) + 1)\n        elif self.config.label_transform == 'quantile':\n            transformer = QuantileTransformer(n_quantiles=min(1000, len(y)), output_distribution='normal')\n            return transformer.fit_transform(y.reshape(-1, 1)).ravel()\n        elif self.config.label_transform == 'log1p':\n            min_val = np.min(y)\n            if min_val < 0:\n                y_shifted = y - min_val + 1\n            else:\n                y_shifted = y\n            return np.log1p(y_shifted)\n        return y\n\nclass ExperimentRunner:\n    def __init__(self, memory_manager: MemoryManager, sheets_tracker: Optional[EnhancedSheetsTracker] = None):\n        self.memory_manager = memory_manager\n        self.sheets_tracker = sheets_tracker\n        self.feature_group_config = Config.FEATURE_GROUP_CONFIG\n        \n    def run_experiment(self, train_df: pd.DataFrame, test_df: pd.DataFrame,\n                      config: ExperimentConfig) -> Dict[str, Any]:\n        start_time = time.time()\n        experiment_id = f\"exp_{datetime.now().strftime('%Y%m%d_%H%M%S')}_{random.randint(1000, 9999)}\"\n        \n        if self.sheets_tracker:\n            success, row_num = self.sheets_tracker.atomic_claim_experiment(experiment_id, config)\n            if not success:\n                print(f\"Configuration already being processed or completed\")\n                return {'success': False, 'reason': 'already_claimed'}\n        \n        print(f\"\\n{'='*60}\")\n        print(f\"Starting experiment {experiment_id}\")\n        print(f\"Config: {config.get_description()}\")\n        print(f\"{'='*60}\")\n        \n        try:\n            if self.sheets_tracker:\n                self.sheets_tracker.update_experiment_status(experiment_id, \"Running\")\n            \n            if not self.memory_manager.check_memory_available(Config.MIN_MEMORY_GB):\n                raise MemoryError(\"Insufficient memory available\")\n            \n            # Prepare data\n            X_train = train_df[config.feature_list]\n            y_train = train_df['label'].values\n            X_test = test_df[config.feature_list]\n            \n            print(f\"Training on FULL dataset: {len(X_train)} samples\")\n            print(f\"Features: {len(config.feature_list)} total\")\n            \n            # Feature engineering\n            engineer = FeatureEngineer(config)\n            engineer.fit(X_train, y_train)\n            \n            X_train_transformed = engineer.transform(X_train)\n            X_test_transformed = engineer.transform(X_test)\n            \n            print(f\"Transformed dimensions: {X_train_transformed.shape[1]}\")\n            \n            # Get feature indices for hierarchical processing\n            feature_indices = self.feature_group_config.get_feature_indices(config.feature_list)\n            \n            # Train model with CV\n            trainer = ModelTrainer(config, self.memory_manager)\n            cv_results = trainer.train_with_cv(X_train_transformed, y_train, feature_indices)\n            \n            print(f\"\\nCV Results:\")\n            print(f\"  Mean train: {cv_results['mean_train_score']:.6f} ± {cv_results['std_train_score']:.6f}\")\n            print(f\"  Mean val: {cv_results['mean_val_score']:.6f} ± {cv_results['std_val_score']:.6f}\")\n            print(f\"  Train-val gap: {cv_results['train_val_gap']:.6f}\")\n            print(f\"  Fold consistency: {cv_results['fold_consistency']:.6f}\")\n            print(f\"  Valid folds: {cv_results['n_valid_folds']}/{Config.N_FOLDS}\")\n            print(f\"  MC Uncertainty: {cv_results['mc_uncertainty']:.6f}\")\n            print(f\"  Overfitting: {cv_results['overfitting_severity']}\")\n            if cv_results['noise_samples_filtered'] > 0:\n                print(f\"  Noise samples filtered: {cv_results['noise_samples_filtered']}\")\n            \n            if self.sheets_tracker:\n                self.sheets_tracker.update_experiment_status(\n                    experiment_id, \n                    \"Generating predictions\",\n                    cv_results\n                )\n            \n            # Generate predictions\n            device = self.memory_manager.get_optimal_device()\n            all_preds = []\n            all_uncertainties = []\n            \n            print(\"\\nGenerating predictions...\")\n            for model_idx, model in enumerate(cv_results['models']):\n                print(f\"  Model {model_idx + 1}/{len(cv_results['models'])}\")\n                model = model.to(device)\n                model.eval()\n                \n                model_preds = []\n                model_uncertainties = []\n                \n                for i in range(0, len(X_test_transformed), Config.MAX_BATCH_SIZE_PREDICT):\n                    batch = torch.FloatTensor(\n                        X_test_transformed[i:i+Config.MAX_BATCH_SIZE_PREDICT]\n                    ).to(device)\n                    \n                    if config.use_mc_dropout:\n                        base_model = model\n                        preds, uncertainty = base_model.mc_forward(\n                            batch, n_samples=Config.MC_DROPOUT_SAMPLES\n                        )\n                        model_preds.extend(preds.detach().cpu().numpy())\n                        model_uncertainties.extend(uncertainty.detach().cpu().numpy())\n                    else:\n                        with torch.no_grad():\n                            preds = model(batch)\n                            model_preds.extend(preds.detach().cpu().numpy())\n                \n                all_preds.append(model_preds)\n                if model_uncertainties:\n                    all_uncertainties.append(model_uncertainties)\n                \n                model = model.cpu()\n                self.memory_manager.clear_memory(device)\n            \n            # Ensemble predictions\n            test_preds = np.mean(all_preds, axis=0)\n            ensemble_uncertainty = np.std(all_preds, axis=0).mean() if len(all_preds) > 1 else 0\n            \n            # Save submission\n            submission_file = f\"submission_{config.get_hash()}.csv\"\n            submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n            submission_df['prediction'] = test_preds\n            submission_df.to_csv(submission_file, index=False)\n            print(f\"Saved predictions to {submission_file}\")\n            \n            # Calculate memory usage\n            gpu_memory_peak = 0\n            if self.memory_manager.gpu_available:\n                for i in range(self.memory_manager.device_count):\n                    gpu_memory_peak = max(\n                        gpu_memory_peak,\n                        torch.cuda.max_memory_allocated(i) / 1e9\n                    )\n            \n            # Prepare results\n            results = {\n                'success': True,\n                'experiment_id': experiment_id,\n                'timestamp_start': datetime.fromtimestamp(start_time).isoformat(),\n                'timestamp_end': datetime.now().isoformat(),\n                'mean_train_score': cv_results['mean_train_score'],\n                'mean_val_score': cv_results['mean_val_score'],\n                'std_train_score': cv_results['std_train_score'],\n                'std_val_score': cv_results['std_val_score'],\n                'train_val_gap': cv_results['train_val_gap'],\n                'best_fold_score': cv_results['best_fold_score'],\n                'worst_fold_score': cv_results['worst_fold_score'],\n                'fold_consistency': cv_results['fold_consistency'],\n                'n_valid_folds': cv_results['n_valid_folds'],\n                'fold_scores': cv_results['fold_scores'],\n                'fold_train_scores': cv_results['fold_train_scores'],\n                'n_folds': cv_results['n_folds'],\n                'training_time_minutes': (time.time() - start_time) / 60,\n                'memory_peak_gb': self.memory_manager.get_memory_info()['cpu_used_gb'],\n                'gpu_memory_peak_gb': gpu_memory_peak,\n                'submission_file': submission_file,\n                'baseline_similarity': 0,\n                'overfitting_flag': cv_results['overfitting_flag'],\n                'overfitting_severity': cv_results['overfitting_severity'],\n                'mc_uncertainty': cv_results['mc_uncertainty'],\n                'ensemble_uncertainty': ensemble_uncertainty,\n                'noise_samples_filtered': cv_results.get('noise_samples_filtered', 0),\n                'status': 'Completed',\n                'worker_id': self.sheets_tracker.worker_id if self.sheets_tracker else 'local'\n            }\n            \n            if self.sheets_tracker:\n                self.sheets_tracker.update_experiment_status(experiment_id, \"Completed\", results)\n                self.sheets_tracker.log_experiment_complete(experiment_id, config, results)\n            \n            # Clean up\n            del cv_results['models']\n            self.memory_manager.clear_memory()\n            \n            print(f\"✓ Experiment completed in {results['training_time_minutes']:.1f} minutes\")\n            \n            return results\n            \n        except Exception as e:\n            error_msg = f\"{type(e).__name__}: {str(e)}\"\n            print(f\"✗ Experiment failed: {error_msg}\")\n            traceback.print_exc()\n            \n            results = {\n                'success': False,\n                'experiment_id': experiment_id,\n                'timestamp_start': datetime.fromtimestamp(start_time).isoformat(),\n                'timestamp_end': datetime.now().isoformat(),\n                'status': 'Error',\n                'error_message': error_msg[:500],\n                'training_time_minutes': (time.time() - start_time) / 60,\n                'worker_id': self.sheets_tracker.worker_id if self.sheets_tracker else 'local'\n            }\n            \n            if self.sheets_tracker:\n                self.sheets_tracker.update_experiment_status(experiment_id, \"Error\", results)\n                self.sheets_tracker.log_experiment_complete(experiment_id, config, results)\n            \n            self.memory_manager.clear_memory()\n            \n            return results\n\nclass RandomSearchConfigGenerator:\n    def __init__(self, core_features: List[str], all_features: List[str]):\n        self.core_features = core_features\n        self.additional_features = [f for f in all_features if f not in core_features and f.startswith('X')]\n        self.additional_features.sort()\n        \n    def generate_random_configs(self, n_configs: int) -> List[ExperimentConfig]:\n        configs = []\n        \n        for _ in range(n_configs):\n            # Random number of additional features\n            n_additional = random.randint(Config.MIN_ADDITIONAL_FEATURES, \n                                        min(Config.MAX_ADDITIONAL_FEATURES, len(self.additional_features)))\n            \n            selected_features = random.sample(self.additional_features, n_additional)\n            \n            feature_list = self.core_features + selected_features\n            \n            # Sample hyperparameters\n            params = self._sample_hyperparameters()\n            \n            # Random regularization options\n            use_spectral_norm = random.random() < 0.3\n            use_gradient_penalty = random.random() < 0.2\n            use_batch_norm = random.random() < 0.3 and not use_spectral_norm\n            use_layer_norm = random.random() < 0.3 and not use_batch_norm\n            use_mc_dropout = random.random() < 0.7\n            use_attention = random.random() < 0.2 and params['architecture']['name'] not in ['very_deep', 'bottleneck']\n            use_residual = random.random() < 0.3\n            \n            use_noise_detection = random.random() < 0.4\n            \n            # Create config\n            config = ExperimentConfig(\n                feature_list=feature_list,\n                additional_features=selected_features,\n                infinity_strategy=params['infinity_strategy'],\n                feature_transforms=params['feature_transforms'],\n                label_transform=params['label_transform'],\n                architecture_name=params['architecture']['name'],\n                architecture_type=params['architecture']['type'],\n                hidden_dims=params['architecture']['hidden_dims'],\n                activation=params['activation'],\n                dropout_rate=params['dropout_rate'],\n                learning_rate=params['learning_rate'],\n                weight_decay=params['weight_decay'],\n                cv_strategy=params['cv_strategy'],\n                optimizer=random.choice(['adam', 'adamw']),\n                gradient_clip=random.choice([0.5, 1.0, 2.0, 5.0]),\n                input_noise=params['input_noise'],\n                dropout_rates_dict=Config.FEATURE_GROUP_CONFIG.dropout_rates,\n                noise_levels_dict=Config.FEATURE_GROUP_CONFIG.noise_levels,\n                lr_multipliers_dict=Config.FEATURE_GROUP_CONFIG.lr_multipliers,\n                use_spectral_norm=use_spectral_norm,\n                use_gradient_penalty=use_gradient_penalty,\n                gradient_penalty_weight=random.choice([0.05, 0.1, 0.2]) if use_gradient_penalty else 0.1,\n                use_batch_norm=use_batch_norm,\n                use_layer_norm=use_layer_norm,\n                use_mc_dropout=use_mc_dropout,\n                use_scheduler=random.choice(['plateau', 'cosine', 'exponential', 'none']),\n                use_attention=use_attention,\n                attention_heads=random.choice([2, 4, 8]) if use_attention else 4,\n                use_residual=use_residual,\n                label_smoothing=random.choice([0.0, 0.01, 0.02, 0.05]),\n                use_noise_detection=use_noise_detection,\n                noise_detection_method='ensemble' if use_noise_detection else 'none'\n            )\n            \n            configs.append(config)\n        \n        return configs\n    \n    def _sample_hyperparameters(self) -> Dict:\n        return {\n            'infinity_strategy': random.choice(Config.INFINITY_STRATEGIES),\n            'feature_transforms': random.choice(Config.FEATURE_TRANSFORMS),\n            'label_transform': random.choice(Config.LABEL_TRANSFORMS),\n            'architecture': random.choice(Config.ARCHITECTURE_VARIANTS),\n            'activation': random.choice(Config.ACTIVATIONS),\n            'dropout_rate': random.choice(Config.DROPOUT_RATES),\n            'learning_rate': random.choice(Config.LEARNING_RATES),\n            'weight_decay': random.choice(Config.WEIGHT_DECAYS),\n            'input_noise': random.choice(Config.INPUT_NOISE_LEVELS),\n            'cv_strategy': random.choice(Config.CV_STRATEGIES)\n        }\n\nclass EnhancedCryptoPipeline:\n    def __init__(self, worker_id: str = None):\n        self.start_time = time.time()\n        self.memory_manager = MemoryManager()\n        self.sheets_tracker = None\n        self.experiment_runner = None\n        self.baseline_config = create_baseline_config()\n        self.baseline_score = None\n        self.worker_id = worker_id\n        \n        if GSPREAD_AVAILABLE and os.path.exists(Config.CREDENTIALS_FILE):\n            try:\n                self.sheets_tracker = EnhancedSheetsTracker(\n                    Config.CREDENTIALS_FILE,\n                    Config.SPREADSHEET_URL,\n                    worker_id=self.worker_id\n                )\n                self.experiment_runner = ExperimentRunner(\n                    self.memory_manager, self.sheets_tracker\n                )\n                print(\"✓ Google Sheets tracking initialized\")\n            except Exception as e:\n                print(f\"✗ Failed to initialize sheets tracker: {e}\")\n                self.experiment_runner = ExperimentRunner(self.memory_manager)\n        else:\n            print(\"Running without Google Sheets tracking\")\n            self.experiment_runner = ExperimentRunner(self.memory_manager)\n    \n    def run(self):\n        print(\"=\"*80)\n        print(\"CRYPTO PREDICTION PIPELINE - REGRESSION V6\")\n        print(f\"Worker ID: {self.sheets_tracker.worker_id if self.sheets_tracker else 'local'}\")\n        print(f\"Available GPUs: {torch.cuda.device_count() if torch.cuda.is_available() else 0}\")\n        print(\"=\"*80)\n        \n        print(\"\\nLoading data...\")\n        train_df = pd.read_parquet(Config.TRAIN_PATH)\n        test_df = pd.read_parquet(Config.TEST_PATH)\n        \n        print(\"Creating microstructure features...\")\n        train_df = create_advanced_microstructure_features(train_df)\n        test_df = create_advanced_microstructure_features(test_df)\n        \n        print(f\"Train shape: {train_df.shape}\")\n        print(f\"Test shape: {test_df.shape}\")\n        \n        # Check for missing features\n        missing_features = [f for f in Config.CORE_FEATURES if f not in train_df.columns]\n        if missing_features:\n            print(f\"Warning: Missing core features: {missing_features}\")\n            Config.CORE_FEATURES = [f for f in Config.CORE_FEATURES if f in train_df.columns]\n        \n        print(f\"Using {len(Config.CORE_FEATURES)} core features\")\n        \n        # Get all features\n        all_features = [col for col in train_df.columns if col not in ['label', 'timestamp']]\n        \n        # Create config generator\n        config_generator = RandomSearchConfigGenerator(Config.CORE_FEATURES, all_features)\n        \n        print(f\"Found {len(config_generator.additional_features)} additional X features\")\n        \n        # Initialize counters\n        n_completed = 0\n        n_errors = 0\n        n_skipped = 0\n        best_score = -np.inf\n        best_config = None\n        \n        # Test baseline model\n        print(\"\\n\" + \"=\"*60)\n        print(\"Testing baseline model\")\n        print(\"=\"*60)\n        \n        baseline_result = self.experiment_runner.run_experiment(\n            train_df, test_df, self.baseline_config\n        )\n        \n        if baseline_result['success']:\n            self.baseline_score = baseline_result['mean_val_score']\n            best_score = self.baseline_score\n            best_config = self.baseline_config\n            n_completed += 1\n            \n            if self.sheets_tracker:\n                self.sheets_tracker.set_baseline_score(self.baseline_score)\n            \n            print(f\"\\n✓ Baseline score: {self.baseline_score:.6f}\")\n        else:\n            if baseline_result.get('reason') != 'already_claimed':\n                print(\"\\n✗ Failed to run baseline model\")\n                self.baseline_score = 0.01\n            else:\n                self.baseline_score = 0.01\n                print(\"Baseline already tested, using default score: 0.01\")\n                n_skipped += 1\n        \n        # Start random search\n        print(\"\\n\" + \"=\"*60)\n        print(\"Starting enhanced random search\")\n        print(\"=\"*60)\n        \n        configs_tried = 0\n        \n        while configs_tried < Config.MAX_CONFIGS_TO_TRY:\n            elapsed_hours = (time.time() - self.start_time) / 3600\n            if elapsed_hours >= Config.MAX_RUNTIME_HOURS:\n                print(f\"\\nReached maximum runtime of {Config.MAX_RUNTIME_HOURS} hours\")\n                break\n            \n            # Generate batch of configs\n            batch_size = min(Config.CONFIGS_PER_FEATURE_SET, Config.MAX_CONFIGS_TO_TRY - configs_tried)\n            random_configs = config_generator.generate_random_configs(batch_size)\n            \n            for config in random_configs:\n                configs_tried += 1\n                \n                print(f\"\\n--- Config {configs_tried}/{Config.MAX_CONFIGS_TO_TRY} ---\")\n                print(f\"Features: {len(config.additional_features)} additional\")\n                print(f\"Architecture: {config.architecture_name}, CV: {config.cv_strategy}\")\n                \n                # Run experiment\n                result = self.experiment_runner.run_experiment(train_df, test_df, config)\n                \n                if result['success']:\n                    n_completed += 1\n                    val_score = result['mean_val_score']\n                    \n                    # Skip if overfitting is severe\n                    if result.get('overfitting_severity') in ['Very High', 'Severe', 'Insufficient Valid Folds']:\n                        print(f\"  Skipping due to {result['overfitting_severity']} overfitting\")\n                        continue\n                    \n                    # Update best score\n                    if val_score > best_score:\n                        best_score = val_score\n                        best_config = config\n                        improvement = best_score - self.baseline_score\n                        print(f\"  🎯 New best score: {best_score:.6f} (improvement: {improvement:.6f})\")\n                        \n                elif result.get('reason') == 'already_claimed':\n                    n_skipped += 1\n                else:\n                    n_errors += 1\n                \n                # Log progress\n                if (n_completed + n_errors + n_skipped) % 10 == 0:\n                    self._log_progress(n_completed, n_errors, n_skipped, best_score)\n                \n                # Check runtime\n                if (time.time() - self.start_time) / 3600 >= Config.MAX_RUNTIME_HOURS:\n                    break\n                \n                # Clear memory periodically\n                if configs_tried % 5 == 0:\n                    self.memory_manager.clear_memory()\n                    print(\"  Cleared memory\")\n        \n        # Save summary\n        self._save_summary(n_completed, n_errors, n_skipped, best_score, best_config)\n    \n    def _log_progress(self, n_completed: int, n_errors: int, n_skipped: int, best_score: float):\n        elapsed_hours = (time.time() - self.start_time) / 3600\n        improvement = best_score - self.baseline_score if self.baseline_score else 0\n        \n        print(f\"\\n--- Progress Update ---\")\n        print(f\"Runtime: {elapsed_hours:.2f} hours\")\n        print(f\"Completed: {n_completed}, Skipped: {n_skipped}, Errors: {n_errors}\")\n        print(f\"Best score: {best_score:.6f} (improvement: {improvement:.6f})\")\n        \n        mem_info = self.memory_manager.get_memory_info()\n        print(f\"Memory: CPU {mem_info['cpu_used_gb']:.1f}GB / {mem_info['cpu_used_gb'] + mem_info['cpu_available_gb']:.1f}GB\")\n        if self.memory_manager.gpu_available:\n            for i in range(self.memory_manager.device_count):\n                if f'gpu_{i}_allocated_gb' in mem_info:\n                    print(f\"  GPU {i}: {mem_info[f'gpu_{i}_allocated_gb']:.1f}GB allocated\")\n        print(\"-\" * 23)\n    \n    def _save_summary(self, n_completed: int, n_errors: int, n_skipped: int, \n                     best_score: float, best_config: Optional[ExperimentConfig]):\n        print(\"\\n\" + \"=\"*80)\n        print(\"PIPELINE COMPLETE\")\n        print(\"=\"*80)\n        print(f\"Total runtime: {(time.time() - self.start_time) / 3600:.2f} hours\")\n        print(f\"Experiments completed: {n_completed}\")\n        print(f\"Experiments skipped: {n_skipped}\")\n        print(f\"Experiments errored: {n_errors}\")\n        print(f\"Baseline score: {self.baseline_score:.6f}\")\n        print(f\"Best validation score: {best_score:.6f}\")\n        print(f\"Improvement: {best_score - self.baseline_score:.6f}\")\n        \n        if best_config:\n            print(f\"\\nBest configuration:\")\n            print(f\"  {best_config.get_description()}\")\n            print(f\"  Additional features ({len(best_config.additional_features)}): {best_config.additional_features[:5]}...\")\n            print(f\"  Architecture: {best_config.architecture_name} with {best_config.activation} activation\")\n            print(f\"  Regularization: Spectral={best_config.use_spectral_norm}, \"\n                  f\"GradPenalty={best_config.use_gradient_penalty}, MCDropout={best_config.use_mc_dropout}\")\n            print(f\"  Advanced features: Attention={best_config.use_attention}, \"\n                  f\"NoiseDetection={best_config.use_noise_detection}\")\n            \n            # Save best config\n            best_config_file = f\"best_config_v6_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json\"\n            with open(best_config_file, 'w') as f:\n                json.dump(best_config.__dict__, f, indent=2, default=str)\n            print(f\"\\nBest configuration saved to: {best_config_file}\")\n\ndef signal_handler(sig, frame):\n    print(\"\\n\\nReceived interrupt signal. Shutting down gracefully...\")\n    sys.exit(0)\n\ndef main():\n    signal.signal(signal.SIGINT, signal_handler)\n    \n    # Set random seeds\n    np.random.seed(42)\n    torch.manual_seed(42)\n    random.seed(42)\n    \n    if torch.cuda.is_available():\n        torch.cuda.manual_seed_all(42)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n        \n        # Set memory growth\n        os.environ['PYTORCH_CUDA_ALLOC_CONF'] = 'expandable_segments:True'\n    \n    # Run pipeline\n    pipeline = EnhancedCryptoPipeline()\n    pipeline.run()\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}