{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":12295803,"sourceType":"datasetVersion","datasetId":7749808}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enhanced Fast Crypto Pipeline v6.0 with Advanced Linear Models\n# Now includes ElasticNet, Lasso, Huber, SVR and feature combinations\n\nimport pandas as pd\nimport numpy as np\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.linear_model import Ridge, ElasticNet, Lasso, HuberRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.neural_network import MLPRegressor\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.stats import pearsonr\nimport warnings\nimport gc\nimport os\nimport json\nimport hashlib\nfrom typing import Dict, List, Tuple, Optional, Any\nfrom datetime import datetime\nimport psutil\nimport traceback\n\nwarnings.filterwarnings('ignore')\n\n# ===== Enhanced Score Persistence System with Feature Tracking =====\nclass ScoreTracker:\n    \"\"\"Persistent score tracking system with feature set tracking\"\"\"\n    def __init__(self, scores_file=\"crypto_scores_history.json\", uploaded_scores_file=None):\n        self.scores_file = scores_file\n        self.scores = {}\n        self.run_history = []\n        self.feature_set_performance = {}  # Track performance by feature set\n        \n        # Load existing scores\n        self.load_scores()\n        \n        # Load uploaded scores if provided\n        if uploaded_scores_file:\n            if isinstance(uploaded_scores_file, list):\n                # Handle multiple files\n                for file in uploaded_scores_file:\n                    if os.path.exists(file):\n                        self.load_uploaded_scores(file)\n            elif os.path.exists(uploaded_scores_file):\n                self.load_uploaded_scores(uploaded_scores_file)\n    \n    def load_scores(self):\n        \"\"\"Load scores from file with error handling\"\"\"\n        if os.path.exists(self.scores_file):\n            try:\n                with open(self.scores_file, 'r') as f:\n                    data = json.load(f)\n                    \n                # Handle different data structures\n                if isinstance(data, dict):\n                    if 'scores' in data:\n                        self.scores = self._migrate_scores(data.get('scores', {}))\n                    else:\n                        # Assume the entire dict is scores\n                        self.scores = self._migrate_scores(data)\n                    self.run_history = data.get('history', [])\n                    self.feature_set_performance = data.get('feature_set_performance', {})\n                elif isinstance(data, list):\n                    # Convert list format to dict\n                    self.scores = self._convert_list_to_dict(data)\n                    \n                print(f\"📚 Loaded {len(self.scores)} previous scores from local file\")\n            except Exception as e:\n                print(f\"⚠️ Error loading scores: {str(e)}\")\n                print(\"Starting with empty score history\")\n    \n    def _migrate_scores(self, scores_dict: Dict) -> Dict:\n        \"\"\"Migrate scores to ensure consistent structure\"\"\"\n        migrated = {}\n        for key, value in scores_dict.items():\n            if isinstance(value, dict):\n                # Ensure cv_score exists\n                if 'cv_score' not in value:\n                    # Try to find score in different fields\n                    if 'score' in value:\n                        value['cv_score'] = value['score']\n                    elif 'validation_score' in value:\n                        value['cv_score'] = value['validation_score']\n                    else:\n                        # Skip entries without scores\n                        continue\n                migrated[key] = value\n            elif isinstance(value, (float, int)):\n                # Convert simple score to dict format\n                migrated[key] = {\n                    'cv_score': float(value),\n                    'timestamp': datetime.now().isoformat()\n                }\n        return migrated\n    \n    def _convert_list_to_dict(self, scores_list: List) -> Dict:\n        \"\"\"Convert list format to dictionary format\"\"\"\n        scores_dict = {}\n        for item in scores_list:\n            if isinstance(item, dict):\n                # Generate hash from item properties\n                config_str = f\"{item.get('model', '')}_{item.get('strategy', '')}_{item.get('param_set', '')}_{item.get('feature_set', '')}\"\n                hash_key = hashlib.md5(config_str.encode()).hexdigest()[:12]\n                scores_dict[hash_key] = item\n        return self._migrate_scores(scores_dict)\n    \n    def load_uploaded_scores(self, filepath):\n        \"\"\"Load scores from uploaded file with better error handling\"\"\"\n        print(f\"📤 Loading scores from: {os.path.basename(filepath)}\")\n        try:\n            if filepath.endswith('.csv'):\n                df = pd.read_csv(filepath)\n                loaded_count = 0\n                # Handle different CSV formats\n                for _, row in df.iterrows():\n                    config_hash = None\n                    cv_score = None\n                    \n                    # Try different column names\n                    if 'config_hash' in row:\n                        config_hash = row['config_hash']\n                    elif 'hash' in row:\n                        config_hash = row['hash']\n                    else:\n                        # Generate hash from model info\n                        config_str = f\"{row.get('model', '')}_{row.get('strategy', '')}_{row.get('param_set', '')}_{row.get('feature_set', '')}\"\n                        config_hash = hashlib.md5(config_str.encode()).hexdigest()[:12]\n                    \n                    # Try different score column names\n                    for score_col in ['cv_score', 'score', 'validation_score', 'val_score']:\n                        if score_col in row and pd.notna(row[score_col]):\n                            cv_score = float(row[score_col])\n                            break\n                    \n                    if config_hash and cv_score is not None:\n                        self.scores[config_hash] = {\n                            'cv_score': cv_score,\n                            'lb_score': row.get('lb_score', None),\n                            'timestamp': row.get('timestamp', datetime.now().isoformat()),\n                            'model_info': {\n                                'model': row.get('model', ''),\n                                'strategy': row.get('strategy', ''),\n                                'param_set': row.get('param_set', 1),\n                                'scaling': row.get('scaling', None),\n                                'feature_set': row.get('feature_set', 'default')\n                            }\n                        }\n                        loaded_count += 1\n                print(f\"  Successfully loaded {loaded_count} scores from CSV\")\n                        \n            elif filepath.endswith('.json'):\n                with open(filepath, 'r') as f:\n                    uploaded = json.load(f)\n                    \n                # Handle different JSON structures\n                if isinstance(uploaded, dict):\n                    if 'scores' in uploaded:\n                        # Nested structure\n                        new_scores = self._migrate_scores(uploaded['scores'])\n                    else:\n                        # Direct scores dict\n                        new_scores = self._migrate_scores(uploaded)\n                    \n                    # Load feature set performance if available\n                    if 'feature_set_performance' in uploaded:\n                        self.feature_set_performance.update(uploaded['feature_set_performance'])\n                elif isinstance(uploaded, list):\n                    # List of scores\n                    new_scores = self._convert_list_to_dict(uploaded)\n                else:\n                    raise ValueError(f\"Unexpected JSON structure: {type(uploaded)}\")\n                \n                # Merge with existing scores\n                self.scores.update(new_scores)\n                print(f\"  Successfully loaded {len(new_scores)} scores from JSON\")\n                    \n            print(f\"📊 Total scores after loading: {len(self.scores)}\")\n            \n        except Exception as e:\n            print(f\"⚠️ Error loading uploaded scores from {os.path.basename(filepath)}: {str(e)}\")\n    \n    def save_scores(self):\n        \"\"\"Save scores to file\"\"\"\n        data = {\n            'scores': self.scores,\n            'history': self.run_history,\n            'feature_set_performance': self.feature_set_performance,\n            'last_updated': datetime.now().isoformat(),\n            'version': '2.1'\n        }\n        with open(self.scores_file, 'w') as f:\n            json.dump(data, f, indent=2)\n    \n    def get_config_hash(self, model_type, strategy, param_set, feature_set, scaling=None):\n        \"\"\"Generate unique hash for configuration including feature set\"\"\"\n        config_str = f\"{model_type}_{strategy}_{param_set}_{feature_set}_{scaling}\"\n        return hashlib.md5(config_str.encode()).hexdigest()[:12]\n    \n    def has_score(self, config_hash):\n        \"\"\"Check if we already have a score for this configuration\"\"\"\n        return config_hash in self.scores\n    \n    def add_score(self, config_hash, cv_score, model_info):\n        \"\"\"Add a new score and update feature set performance\"\"\"\n        self.scores[config_hash] = {\n            'cv_score': float(cv_score),\n            'model_info': model_info,\n            'timestamp': datetime.now().isoformat()\n        }\n        \n        # Update feature set performance tracking\n        feature_set = model_info.get('feature_set', 'default')\n        if feature_set not in self.feature_set_performance:\n            self.feature_set_performance[feature_set] = {\n                'scores': [],\n                'avg_score': 0,\n                'best_score': 0,\n                'model_count': 0\n            }\n        \n        self.feature_set_performance[feature_set]['scores'].append(cv_score)\n        self.feature_set_performance[feature_set]['avg_score'] = np.mean(self.feature_set_performance[feature_set]['scores'])\n        self.feature_set_performance[feature_set]['best_score'] = max(self.feature_set_performance[feature_set]['scores'])\n        self.feature_set_performance[feature_set]['model_count'] = len(self.feature_set_performance[feature_set]['scores'])\n        \n        self.save_scores()\n    \n    def get_best_configs(self, n=10):\n        \"\"\"Get best performing configurations with error handling\"\"\"\n        # Filter out entries without cv_score\n        valid_scores = {k: v for k, v in self.scores.items() \n                       if isinstance(v, dict) and 'cv_score' in v and v['cv_score'] is not None}\n        \n        if not valid_scores:\n            return []\n        \n        sorted_scores = sorted(valid_scores.items(), \n                             key=lambda x: float(x[1]['cv_score']), \n                             reverse=True)\n        return sorted_scores[:n]\n    \n    def get_best_by_feature_set(self, n=3):\n        \"\"\"Get best models for each feature set\"\"\"\n        feature_set_best = {}\n        \n        for config_hash, score_data in self.scores.items():\n            if isinstance(score_data, dict) and 'cv_score' in score_data:\n                model_info = score_data.get('model_info', {})\n                feature_set = model_info.get('feature_set', 'default')\n                \n                if feature_set not in feature_set_best:\n                    feature_set_best[feature_set] = []\n                \n                feature_set_best[feature_set].append((config_hash, score_data))\n        \n        # Sort each feature set's models\n        for feature_set in feature_set_best:\n            feature_set_best[feature_set] = sorted(\n                feature_set_best[feature_set],\n                key=lambda x: x[1]['cv_score'],\n                reverse=True\n            )[:n]\n        \n        return feature_set_best\n    \n    def add_run_summary(self, summary):\n        \"\"\"Add summary of current run\"\"\"\n        self.run_history.append({\n            'timestamp': datetime.now().isoformat(),\n            'summary': summary\n        })\n        self.save_scores()\n    \n    def export_to_csv(self, output_file=\"scores_export.csv\"):\n        \"\"\"Export scores to CSV for easy viewing and re-upload\"\"\"\n        scores_data = []\n        for config_hash, score_data in self.scores.items():\n            if isinstance(score_data, dict):\n                model_info = score_data.get('model_info', {})\n                row = {\n                    'config_hash': config_hash,\n                    'cv_score': score_data.get('cv_score', None),\n                    'lb_score': score_data.get('lb_score', None),\n                    'timestamp': score_data.get('timestamp', ''),\n                    'model': model_info.get('model', '') if isinstance(model_info, dict) else '',\n                    'strategy': model_info.get('strategy', '') if isinstance(model_info, dict) else '',\n                    'param_set': model_info.get('param_set', '') if isinstance(model_info, dict) else '',\n                    'scaling': model_info.get('scaling', '') if isinstance(model_info, dict) else '',\n                    'feature_set': model_info.get('feature_set', 'default') if isinstance(model_info, dict) else 'default'\n                }\n                scores_data.append(row)\n        \n        if scores_data:\n            df = pd.DataFrame(scores_data)\n            df = df.sort_values('cv_score', ascending=False)\n            df.to_csv(output_file, index=False)\n            print(f\"📥 Scores exported to: {output_file}\")\n            return output_file\n        else:\n            print(\"⚠️ No scores to export\")\n            return None\n    \n    def print_feature_set_summary(self):\n        \"\"\"Print summary of feature set performance\"\"\"\n        if not self.feature_set_performance:\n            print(\"No feature set performance data available\")\n            return\n        \n        print(\"\\n📊 FEATURE SET PERFORMANCE SUMMARY:\")\n        print(\"=\" * 80)\n        \n        # Sort by best score\n        sorted_sets = sorted(\n            self.feature_set_performance.items(),\n            key=lambda x: x[1]['best_score'],\n            reverse=True\n        )\n        \n        for feature_set, stats in sorted_sets:\n            print(f\"\\n{feature_set}:\")\n            print(f\"  Models trained: {stats['model_count']}\")\n            print(f\"  Best CV score: {stats['best_score']:.4f}\")\n            print(f\"  Average CV score: {stats['avg_score']:.4f}\")\n\n# ===== Enhanced Configuration with Feature Sets =====\nclass FastConfig:\n    \"\"\"Optimized configuration with feature set variations\"\"\"\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # Scores tracking - now supports multiple files\n    SCORES_FILE = \"crypto_scores_history.json\"\n    UPLOADED_SCORES_PATHS = [\n        \"/kaggle/input/1-set-of-test-records-drw/crypto_scores_history.json\",\n        # Add more paths here as needed\n        # \"/kaggle/input/crypto-scores-2/scores_export.csv\",\n        # \"/kaggle/input/run-results/crypto_scores_history.json\",\n    ]\n    \n    # Define different feature sets to experiment with\n    FEATURE_SETS = {\n        \"top_features\": {\n            \"name\": \"Top Features Only\",\n            \"features\": [\n                \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\",\n                \"buy_qty\", \"sell_qty\", \"volume\", \"bid_qty\", \"ask_qty\"\n            ],\n            \"engineered\": [\n                \"buy_sell_ratio\", \"volume_weighted_sell\", \"bid_ask_imbalance\",\n                \"order_flow_imbalance\", \"log_volume\"\n            ]\n        },\n        \"extended_v1\": {\n            \"name\": \"Extended Features V1\",\n            \"features\": [\n                # Original top features\n                \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\",\n                \"buy_qty\", \"sell_qty\", \"volume\", \"bid_qty\", \"ask_qty\",\n                # Additional high-correlation features\n                \"X858\", \"X601\", \"X602\", \"X849\", \"X382\", \"X853\", \"X607\", \"X855\"\n            ],\n            \"engineered\": [\n                \"buy_sell_ratio\", \"volume_weighted_sell\", \"bid_ask_imbalance\",\n                \"order_flow_imbalance\", \"log_volume\",\n                # New engineered features\n                \"volume_per_trade\", \"bid_ask_spread\", \"liquidity_imbalance\"\n            ]\n        },\n        \"market_micro\": {\n            \"name\": \"Market Microstructure Focus\",\n            \"features\": [\n                \"buy_qty\", \"sell_qty\", \"volume\", \"bid_qty\", \"ask_qty\",\n                # Add some X features that might capture microstructure\n                \"X863\", \"X856\", \"X598\", \"X862\", \"X860\"\n            ],\n            \"engineered\": [\n                \"buy_sell_ratio\", \"bid_ask_imbalance\", \"order_flow_imbalance\",\n                \"volume_weighted_sell\", \"log_volume\",\n                \"trade_intensity\", \"liquidity_ratio\", \"pressure_indicator\"\n            ]\n        },\n        \"minimal\": {\n            \"name\": \"Minimal Feature Set\",\n            \"features\": [\n                # Only the most essential features\n                \"X863\", \"X856\", \"buy_qty\", \"sell_qty\", \"volume\"\n            ],\n            \"engineered\": [\n                \"buy_sell_ratio\", \"log_volume\"\n            ]\n        },\n        \"x_features_only\": {\n            \"name\": \"X Features Only\",\n            \"features\": [\n                # Focus on anonymous features only\n                \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\",\n                \"X858\", \"X601\", \"X602\", \"X849\", \"X382\", \"X853\", \"X607\", \"X855\",\n                \"X859\", \"X604\", \"X606\", \"X850\", \"X381\", \"X854\", \"X608\", \"X857\"\n            ],\n            \"engineered\": []  # No engineered features for this set\n        },\n        \"x_features_combinations\": {\n            \"name\": \"X Features with Combinations\",\n            \"features\": [\n                # Top 8 X features\n                \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\"\n            ],\n            \"engineered\": [\n                # Feature combinations - these will be created in feature engineering\n                \"X863_X856_product\",\n                \"X598_X862_product\", \n                \"X863_X856_ratio\",\n                \"X598_X862_ratio\",\n                \"X385_squared\",\n                \"X852_squared\",\n                \"X863_log\",\n                \"X856_log\",\n                \"X863_X860_sum\",\n                \"X856_X852_diff\"\n            ]\n        }\n    }\n    \n    LABEL_COLUMN = \"label\"\n    RANDOM_STATE = 42\n    \n    # Speed optimizations\n    FAST_MODE = True\n    MIN_CV_THRESHOLD = 0.05  # Skip models below this CV\n    MAX_MODELS_PER_TYPE = 3  # Limit variations per model type\n    USE_PARALLEL = False  # Set True if you have multiple CPUs\n    N_JOBS = 2  # Number of parallel jobs\n    \n    # Ensemble settings\n    MIN_ENSEMBLE_SIZE = 3\n    MAX_ENSEMBLE_SIZE = 10\n\n# ===== Memory monitoring =====\ndef get_memory_usage():\n    \"\"\"Get current memory usage in MB\"\"\"\n    process = psutil.Process(os.getpid())\n    return process.memory_info().rss / 1024 / 1024\n\ndef log_memory(message=\"\"):\n    \"\"\"Log current memory usage\"\"\"\n    print(f\"💾 Memory: {get_memory_usage():.0f}MB {message}\")\n\n# ===== Enhanced Feature Engineering =====\ndef fast_feature_engineering(df: pd.DataFrame, feature_set_config: Dict) -> pd.DataFrame:\n    \"\"\"Optimized feature engineering with configurable feature sets\"\"\"\n    # Get features for this configuration\n    base_features = feature_set_config[\"features\"]\n    engineered_features = feature_set_config.get(\"engineered\", [])\n    \n    # Only keep necessary columns\n    required_cols = base_features + [FastConfig.LABEL_COLUMN] if FastConfig.LABEL_COLUMN in df.columns else base_features\n    \n    # Filter to only existing columns\n    existing_cols = [col for col in required_cols if col in df.columns]\n    df = df[existing_cols].copy()\n    \n    # Create engineered features based on configuration\n    if \"buy_sell_ratio\" in engineered_features and all(col in df.columns for col in [\"buy_qty\", \"sell_qty\"]):\n        df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    \n    if \"volume_weighted_sell\" in engineered_features and all(col in df.columns for col in [\"sell_qty\", \"volume\"]):\n        df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    \n    if \"bid_ask_imbalance\" in engineered_features and all(col in df.columns for col in [\"bid_qty\", \"ask_qty\"]):\n        df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-8)\n    \n    if \"order_flow_imbalance\" in engineered_features and all(col in df.columns for col in [\"buy_qty\", \"sell_qty\"]):\n        df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    \n    if \"log_volume\" in engineered_features and \"volume\" in df.columns:\n        df['log_volume'] = np.log1p(df['volume'])\n    \n    # New engineered features\n    if \"volume_per_trade\" in engineered_features and all(col in df.columns for col in [\"volume\", \"buy_qty\", \"sell_qty\"]):\n        df['volume_per_trade'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    \n    if \"bid_ask_spread\" in engineered_features and all(col in df.columns for col in [\"bid_qty\", \"ask_qty\"]):\n        df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    if \"liquidity_imbalance\" in engineered_features and all(col in df.columns for col in [\"bid_qty\", \"ask_qty\", \"volume\"]):\n        df['liquidity_imbalance'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1)\n    \n    if \"trade_intensity\" in engineered_features and all(col in df.columns for col in [\"buy_qty\", \"sell_qty\"]):\n        df['trade_intensity'] = df['buy_qty'] + df['sell_qty']\n    \n    if \"liquidity_ratio\" in engineered_features and all(col in df.columns for col in [\"bid_qty\", \"ask_qty\", \"volume\"]):\n        df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-8)\n    \n    if \"pressure_indicator\" in engineered_features and all(col in df.columns for col in [\"buy_qty\", \"sell_qty\", \"bid_qty\", \"ask_qty\"]):\n        df['pressure_indicator'] = (df['buy_qty'] * df['bid_qty']) / (df['sell_qty'] * df['ask_qty'] + 1e-8)\n    \n    # Feature combinations for x_features_combinations\n    if \"X863_X856_product\" in engineered_features and all(col in df.columns for col in [\"X863\", \"X856\"]):\n        df['X863_X856_product'] = df['X863'] * df['X856']\n    \n    if \"X598_X862_product\" in engineered_features and all(col in df.columns for col in [\"X598\", \"X862\"]):\n        df['X598_X862_product'] = df['X598'] * df['X862']\n    \n    if \"X863_X856_ratio\" in engineered_features and all(col in df.columns for col in [\"X863\", \"X856\"]):\n        df['X863_X856_ratio'] = df['X863'] / (df['X856'] + 1e-8)\n    \n    if \"X598_X862_ratio\" in engineered_features and all(col in df.columns for col in [\"X598\", \"X862\"]):\n        df['X598_X862_ratio'] = df['X598'] / (df['X862'] + 1e-8)\n    \n    if \"X385_squared\" in engineered_features and \"X385\" in df.columns:\n        df['X385_squared'] = df['X385'] ** 2\n    \n    if \"X852_squared\" in engineered_features and \"X852\" in df.columns:\n        df['X852_squared'] = df['X852'] ** 2\n    \n    if \"X863_log\" in engineered_features and \"X863\" in df.columns:\n        df['X863_log'] = np.log1p(np.abs(df['X863']))\n    \n    if \"X856_log\" in engineered_features and \"X856\" in df.columns:\n        df['X856_log'] = np.log1p(np.abs(df['X856']))\n    \n    if \"X863_X860_sum\" in engineered_features and all(col in df.columns for col in [\"X863\", \"X860\"]):\n        df['X863_X860_sum'] = df['X863'] + df['X860']\n    \n    if \"X856_X852_diff\" in engineered_features and all(col in df.columns for col in [\"X856\", \"X852\"]):\n        df['X856_X852_diff'] = df['X856'] - df['X852']\n    \n    # Handle infinities and NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(df.median())\n    \n    return df\n\n# ===== Get feature list for model =====\ndef get_feature_list(feature_set_config: Dict) -> List[str]:\n    \"\"\"Get complete feature list including engineered features\"\"\"\n    base_features = feature_set_config[\"features\"]\n    engineered_features = feature_set_config.get(\"engineered\", [])\n    return base_features + engineered_features\n\n# ===== Updated model parameters with advanced linear models =====\ndef get_fast_model_params(model_type: str, param_set: int = 1) -> Dict:\n    \"\"\"Get optimized parameters for faster training\"\"\"\n    \n    # Reduced iterations/estimators for speed\n    if model_type == \"xgb\":\n        return {\n            \"n_estimators\": 200,  # Reduced from 500\n            \"learning_rate\": 0.05,\n            \"max_depth\": 8,\n            \"subsample\": 0.8,\n            \"colsample_bytree\": 0.8,\n            \"tree_method\": \"hist\",\n            \"random_state\": FastConfig.RANDOM_STATE\n        }\n    \n    elif model_type == \"lgbm\":\n        return {\n            \"n_estimators\": 200,\n            \"learning_rate\": 0.05,\n            \"num_leaves\": 31,\n            \"subsample\": 0.8,\n            \"colsample_bytree\": 0.8,\n            \"device\": \"cpu\",\n            \"random_state\": FastConfig.RANDOM_STATE,\n            \"verbosity\": -1\n        }\n    \n    elif model_type == \"ridge\":\n        return {\"alpha\": 1.0, \"random_state\": FastConfig.RANDOM_STATE}\n    \n    elif model_type == \"elasticnet\":\n        return {\n            \"alpha\": 0.001,\n            \"l1_ratio\": 0.5,\n            \"random_state\": FastConfig.RANDOM_STATE,\n            \"max_iter\": 1000\n        }\n    \n    elif model_type == \"lasso\":\n        return {\n            \"alpha\": 0.0001,\n            \"random_state\": FastConfig.RANDOM_STATE,\n            \"max_iter\": 1000\n        }\n    \n    elif model_type == \"huber\":\n        return {\n            \"epsilon\": 1.35,  # Robust to outliers\n            \"alpha\": 0.0001,\n            \"max_iter\": 100\n        }\n    \n    elif model_type == \"svr_linear\":\n        return {\n            \"kernel\": \"linear\",\n            \"C\": 1.0,\n            \"epsilon\": 0.1,\n            \"max_iter\": 1000\n        }\n    \n    elif model_type == \"nn_simple\":\n        return {\n            \"hidden_layer_sizes\": (64, 32),  # Smaller network\n            \"activation\": \"relu\",\n            \"solver\": \"adam\",\n            \"alpha\": 0.001,\n            \"learning_rate_init\": 0.001,\n            \"max_iter\": 200,  # Reduced iterations\n            \"early_stopping\": True,\n            \"validation_fraction\": 0.1,\n            \"n_iter_no_change\": 10,\n            \"random_state\": FastConfig.RANDOM_STATE\n        }\n    \n    return {}\n\ndef get_priority_strategies() -> List[Dict]:\n    \"\"\"Get only high-performing strategies with granular recent data\"\"\"\n    strategies = []\n    \n    # Original strategies\n    strategies.extend([\n        {\"name\": \"last_75pct\", \"start_pct\": 25, \"end_pct\": 100},\n        {\"name\": \"last_60pct\", \"start_pct\": 40, \"end_pct\": 100},\n        {\"name\": \"last_50pct\", \"start_pct\": 50, \"end_pct\": 100},\n        {\"name\": \"recent_3months\", \"start_pct\": 75, \"end_pct\": 100},\n        {\"name\": \"full_data\", \"start_pct\": 0, \"end_pct\": 100},\n    ])\n    \n    # Granular recent data strategies (90-99%)\n    for pct in range(90, 100):\n        strategies.append({\n            \"name\": f\"last_{pct}pct\",\n            \"start_pct\": 100 - pct,\n            \"end_pct\": 100\n        })\n    \n    return strategies\n\ndef train_model_fast(train_df: pd.DataFrame, test_df: pd.DataFrame, strategy: Dict, \n                    model_type: str, feature_set_config: Dict, feature_set_name: str,\n                    param_set: int = 1, scaling: str = None) -> Tuple[Optional[np.ndarray], Optional[float]]:\n    \"\"\"Optimized model training with feature set support\"\"\"\n    \n    # Get strategy data\n    n_samples = len(train_df)\n    start_idx = int(strategy.get(\"start_pct\", 0) / 100 * n_samples)\n    end_idx = int(strategy.get(\"end_pct\", 100) / 100 * n_samples)\n    strategy_data = train_df.iloc[start_idx:end_idx].reset_index(drop=True)\n    \n    if len(strategy_data) < 10000:\n        return None, None\n    \n    # Split data\n    split_idx = int(0.8 * len(strategy_data))\n    train_data = strategy_data.iloc[:split_idx]\n    valid_data = strategy_data.iloc[split_idx:]\n    \n    # Clear strategy_data\n    del strategy_data\n    gc.collect()\n    \n    # Get features for this configuration\n    features = get_feature_list(feature_set_config)\n    \n    # Filter to only existing features\n    existing_features = [f for f in features if f in train_data.columns]\n    \n    if len(existing_features) < 3:  # Need at least some features\n        print(f\"  Warning: Only {len(existing_features)} features available for {feature_set_name}\")\n        return None, None\n    \n    # Prepare data\n    X_train = train_data[existing_features].values\n    y_train = train_data[FastConfig.LABEL_COLUMN].values\n    X_valid = valid_data[existing_features].values\n    y_valid = valid_data[FastConfig.LABEL_COLUMN].values\n    X_test = test_df[existing_features].values\n    \n    # Clear dataframes\n    del train_data, valid_data\n    gc.collect()\n    \n    # Apply scaling for models that need it\n    scaler = None\n    if model_type in [\"nn_simple\", \"ridge\", \"elasticnet\", \"lasso\", \"huber\", \"svr_linear\"]:\n        scaler = StandardScaler()\n        X_train = scaler.fit_transform(X_train)\n        X_valid = scaler.transform(X_valid)\n        X_test = scaler.transform(X_test)\n    \n    # Get model parameters\n    params = get_fast_model_params(model_type, param_set)\n    if not params:\n        return None, None\n    \n    try:\n        if model_type == \"xgb\":\n            model = XGBRegressor(**params)\n            model.fit(\n                X_train, y_train,\n                eval_set=[(X_valid, y_valid)],\n                early_stopping_rounds=20,\n                verbose=False\n            )\n        elif model_type == \"lgbm\":\n            model = LGBMRegressor(**params)\n            model.fit(\n                X_train, y_train,\n                eval_set=[(X_valid, y_valid)],\n                callbacks=[lambda x: None]\n            )\n        elif model_type == \"ridge\":\n            model = Ridge(**params)\n            model.fit(X_train, y_train)\n        elif model_type == \"elasticnet\":\n            model = ElasticNet(**params)\n            model.fit(X_train, y_train)\n        elif model_type == \"lasso\":\n            model = Lasso(**params)\n            model.fit(X_train, y_train)\n        elif model_type == \"huber\":\n            model = HuberRegressor(**params)\n            model.fit(X_train, y_train)\n        elif model_type == \"svr_linear\":\n            model = SVR(**params)\n            model.fit(X_train, y_train)\n        elif model_type == \"nn_simple\":\n            model = MLPRegressor(**params)\n            model.fit(X_train, y_train)\n        else:\n            return None, None\n        \n        # Predict\n        valid_pred = model.predict(X_valid)\n        test_pred = model.predict(X_test)\n        \n        # Calculate score\n        score = pearsonr(y_valid, valid_pred)[0]\n        if np.isnan(score):\n            score = 0.0\n        \n        # Clear model and data\n        del model, X_train, y_train, X_valid, y_valid, valid_pred\n        gc.collect()\n        \n        return test_pred, score\n        \n    except Exception as e:\n        print(f\"  Error: {str(e)[:50]}...\")\n        return None, None\n\ndef create_fast_ensembles(predictions_cache: Dict, submission_df: pd.DataFrame, \n                         model_results: List[Dict], score_tracker: ScoreTracker) -> List[str]:\n    \"\"\"Create only best ensemble types\"\"\"\n    print(\"\\n🎯 Creating Fast Ensembles...\")\n    ensemble_files = []\n    \n    # Sort by CV score\n    sorted_models = sorted(model_results, key=lambda x: x['cv_score'], reverse=True)\n    \n    # Only create top-performing ensemble types\n    ensemble_configs = [\n        {\"size\": 3, \"method\": \"simple\"},\n        {\"size\": 5, \"method\": \"simple\"},\n        {\"size\": 7, \"method\": \"simple\"},\n        {\"size\": 5, \"method\": \"weighted\"},\n    ]\n    \n    for config in ensemble_configs:\n        size = config[\"size\"]\n        method = config[\"method\"]\n        \n        if len(sorted_models) >= size:\n            selected = sorted_models[:size]\n            predictions = []\n            weights = []\n            \n            for model in selected:\n                if model['filename'] in predictions_cache:\n                    predictions.append(predictions_cache[model['filename']])\n                    if method == \"weighted\":\n                        weights.append(model['cv_score'])\n            \n            if len(predictions) >= size - 1:\n                if method == \"simple\":\n                    ensemble_pred = np.mean(predictions, axis=0)\n                    ensemble_cv = np.mean([m['cv_score'] for m in selected[:len(predictions)]])\n                else:  # weighted\n                    weights = np.array(weights)\n                    weights = weights / weights.sum()\n                    ensemble_pred = np.average(predictions, axis=0, weights=weights)\n                    ensemble_cv = np.average([m['cv_score'] for m in selected[:len(predictions)]], weights=weights)\n                \n                filename = f\"ensemble_{method}_top{size}_cv{ensemble_cv:.4f}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv\"\n                \n                submission = submission_df.copy()\n                submission['prediction'] = ensemble_pred\n                submission.to_csv(filename, index=False)\n                \n                ensemble_files.append(filename)\n                print(f\"  ✅ {method.capitalize()} Top {size} (CV: {ensemble_cv:.4f})\")\n                \n                # Track ensemble\n                config_hash = f\"ensemble_{method}_{size}\"\n                score_tracker.add_score(config_hash, ensemble_cv, {\n                    \"type\": \"ensemble\",\n                    \"method\": method,\n                    \"size\": size,\n                    \"feature_set\": \"mixed\"  # Ensembles can mix feature sets\n                })\n    \n    return ensemble_files\n\n# ===== Main Pipeline with Feature Set Support =====\ndef run_fast_pipeline(uploaded_scores_files=None, feature_sets_to_test=None):\n    \"\"\"Run optimized pipeline with feature set tracking\"\"\"\n    \n    print(\"\\n🚀 FAST CRYPTO PIPELINE v6.0 WITH ADVANCED LINEAR MODELS\")\n    print(\"=\" * 80)\n    log_memory(\"Starting\")\n    \n    # Handle multiple score files\n    if uploaded_scores_files is None:\n        # Check all configured paths\n        uploaded_scores_files = []\n        for path in FastConfig.UPLOADED_SCORES_PATHS:\n            if os.path.exists(path):\n                uploaded_scores_files.append(path)\n                print(f\"✅ Found score file: {os.path.basename(path)}\")\n    \n    # Initialize score tracker\n    score_tracker = ScoreTracker(\n        scores_file=FastConfig.SCORES_FILE,\n        uploaded_scores_file=uploaded_scores_files if uploaded_scores_files else None\n    )\n    \n    # Debug scores structure if needed\n    if len(score_tracker.scores) > 0:\n        print(f\"\\n📊 Loaded {len(score_tracker.scores)} historical scores\")\n    \n    # Show previous best scores (with error handling)\n    try:\n        best_configs = score_tracker.get_best_configs(5)\n        if best_configs:\n            print(\"\\n🏆 Previous Best Scores:\")\n            for i, (config_hash, score_data) in enumerate(best_configs, 1):\n                model_info = score_data.get('model_info', {})\n                if isinstance(model_info, dict):\n                    model_type = model_info.get('model', model_info.get('type', 'Unknown'))\n                    strategy = model_info.get('strategy', model_info.get('method', ''))\n                    feature_set = model_info.get('feature_set', 'default')\n                else:\n                    model_type = 'Unknown'\n                    strategy = ''\n                    feature_set = 'default'\n                cv_score = score_data.get('cv_score', 0)\n                print(f\"  {i}. {model_type:10} {strategy:15} {feature_set:15} CV: {cv_score:.4f}\")\n    except Exception as e:\n        print(f\"⚠️ Could not display previous scores: {str(e)}\")\n    \n    # Show feature set performance summary\n    score_tracker.print_feature_set_summary()\n    \n    # Determine which feature sets to test\n    if feature_sets_to_test is None:\n        feature_sets_to_test = list(FastConfig.FEATURE_SETS.keys())\n    \n    print(f\"\\n🔬 Testing {len(feature_sets_to_test)} feature sets: {', '.join(feature_sets_to_test)}\")\n    \n    # Load data with all possible columns\n    all_features = set()\n    for fs_config in FastConfig.FEATURE_SETS.values():\n        all_features.update(fs_config[\"features\"])\n    all_features = list(all_features)\n    \n    print(\"\\n📂 Loading data...\")\n    train_df = pd.read_parquet(FastConfig.TRAIN_PATH, \n                              columns=all_features + [FastConfig.LABEL_COLUMN])\n    test_df = pd.read_parquet(FastConfig.TEST_PATH, \n                             columns=all_features)\n    submission_df = pd.read_csv(FastConfig.SUBMISSION_PATH)\n    \n    print(f\"  Loaded {len(all_features)} unique features\")\n    \n    # Get configurations\n    strategies = get_priority_strategies()\n    \n    # Priority model configurations including advanced linear models\n    model_configs = [\n        {\"model\": \"ridge\", \"param_set\": 1, \"scaling\": \"standard\"},\n        {\"model\": \"elasticnet\", \"param_set\": 1, \"scaling\": \"standard\"},\n        {\"model\": \"lasso\", \"param_set\": 1, \"scaling\": \"standard\"},\n        {\"model\": \"huber\", \"param_set\": 1, \"scaling\": \"standard\"},\n        {\"model\": \"svr_linear\", \"param_set\": 1, \"scaling\": \"standard\"},\n        {\"model\": \"lgbm\", \"param_set\": 1},\n        {\"model\": \"xgb\", \"param_set\": 1},\n        {\"model\": \"nn_simple\", \"param_set\": 1, \"scaling\": \"standard\"},\n    ]\n    \n    # Create combinations but skip already computed ones\n    all_configs = []\n    skipped_count = 0\n    reused_scores = []\n    \n    # Process each feature set\n    processed_dfs = {}  # Cache processed dataframes\n    \n    for feature_set_name in feature_sets_to_test:\n        feature_set_config = FastConfig.FEATURE_SETS[feature_set_name]\n        \n        print(f\"\\n🔧 Processing feature set: {feature_set_name}\")\n        \n        # Apply feature engineering for this feature set\n        train_df_fs = fast_feature_engineering(train_df.copy(), feature_set_config)\n        test_df_fs = fast_feature_engineering(test_df.copy(), feature_set_config)\n        \n        processed_dfs[feature_set_name] = {\n            'train': train_df_fs,\n            'test': test_df_fs,\n            'config': feature_set_config\n        }\n        \n        print(f\"  Features available: {len(get_feature_list(feature_set_config))}\")\n        \n        for strategy in strategies:\n            for model_config in model_configs:\n                config = model_config.copy()\n                config[\"strategy\"] = strategy[\"name\"]\n                config[\"strategy_dict\"] = strategy\n                config[\"feature_set\"] = feature_set_name\n                config[\"feature_set_config\"] = feature_set_config\n                \n                # Check if already computed\n                config_hash = score_tracker.get_config_hash(\n                    model_config[\"model\"], \n                    strategy[\"name\"], \n                    model_config.get(\"param_set\", 1),\n                    feature_set_name,\n                    model_config.get(\"scaling\", None)\n                )\n                \n                if score_tracker.has_score(config_hash):\n                    # Reuse existing score\n                    score_data = score_tracker.scores[config_hash]\n                    if 'cv_score' in score_data:\n                        reused_scores.append({\n                            'model': model_config[\"model\"],\n                            'strategy': strategy[\"name\"],\n                            'feature_set': feature_set_name,\n                            'cv_score': score_data['cv_score'],\n                            'config_hash': config_hash\n                        })\n                        skipped_count += 1\n                        continue\n                \n                config[\"config_hash\"] = config_hash\n                all_configs.append(config)\n    \n    print(f\"\\n📊 Configuration Summary:\")\n    print(f\"  - Total possible: {len(feature_sets_to_test) * len(strategies) * len(model_configs)}\")\n    print(f\"  - Already computed: {skipped_count}\")\n    print(f\"  - To compute: {len(all_configs)}\")\n    \n    if reused_scores:\n        print(f\"\\n♻️ Reusing {len(reused_scores)} previous scores:\")\n        # Show best reused scores by feature set\n        fs_groups = {}\n        for score in reused_scores:\n            fs = score['feature_set']\n            if fs not in fs_groups:\n                fs_groups[fs] = []\n            fs_groups[fs].append(score)\n        \n        for fs, scores in fs_groups.items():\n            best = sorted(scores, key=lambda x: x['cv_score'], reverse=True)[0]\n            print(f\"  {fs:20} - Best: {best['model']:10} {best['strategy']:15} CV: {best['cv_score']:.4f}\")\n    \n    submissions_created = []\n    model_results = []\n    predictions_cache = {}\n    \n    # Include reused scores in model results for ensemble creation\n    for reused in reused_scores:\n        model_results.append({\n            'filename': f\"reused_{reused['model']}_{reused['strategy']}_{reused['feature_set']}_cv{reused['cv_score']:.4f}\",\n            'cv_score': reused['cv_score'],\n            'model_type': reused['model'],\n            'temporal_strategy': reused['strategy'],\n            'feature_set': reused['feature_set'],\n            'reused': True\n        })\n    \n    # Track progress\n    start_time = datetime.now()\n    completed = 0\n    \n    if len(all_configs) > 0:\n        print(f\"\\n🏃 Running {len(all_configs)} new model configurations...\")\n        \n        for idx, config in enumerate(all_configs, 1):\n            model_type = config['model']\n            param_set = config.get('param_set', 1)\n            scaling = config.get('scaling', None)\n            strategy_name = config['strategy']\n            strategy = config['strategy_dict']\n            feature_set_name = config['feature_set']\n            feature_set_config = config['feature_set_config']\n            config_hash = config['config_hash']\n            \n            print(f\"\\n[{idx}/{len(all_configs)}] {model_type} - {strategy_name} - {feature_set_name}\", end=\"\")\n            if scaling:\n                print(f\" ({scaling})\", end=\"\")\n            \n            # Get preprocessed data for this feature set\n            train_df_fs = processed_dfs[feature_set_name]['train']\n            test_df_fs = processed_dfs[feature_set_name]['test']\n            \n            # Train model\n            test_pred, cv_score = train_model_fast(\n                train_df_fs, test_df_fs, strategy, model_type, \n                feature_set_config, feature_set_name,\n                param_set, scaling\n            )\n            \n            if test_pred is not None and cv_score is not None:\n                # Skip if below threshold\n                if cv_score < FastConfig.MIN_CV_THRESHOLD:\n                    print(f\" → CV: {cv_score:.4f} ❌ (Below threshold)\")\n                    continue\n                \n                # Generate filename\n                filename = f\"sub_{model_type}_{strategy_name}_{feature_set_name}_cv{cv_score:.4f}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv\"\n                \n                # Save submission\n                submission = submission_df.copy()\n                submission['prediction'] = test_pred\n                submission.to_csv(filename, index=False)\n                \n                # Cache prediction\n                predictions_cache[filename] = test_pred\n                \n                # Track score\n                model_info = {\n                    \"model\": model_type,\n                    \"strategy\": strategy_name,\n                    \"param_set\": param_set,\n                    \"scaling\": scaling,\n                    \"feature_set\": feature_set_name,\n                    \"filename\": filename\n                }\n                score_tracker.add_score(config_hash, cv_score, model_info)\n                \n                submissions_created.append(filename)\n                model_results.append({\n                    'filename': filename,\n                    'cv_score': cv_score,\n                    'model_type': model_type,\n                    'temporal_strategy': strategy_name,\n                    'feature_set': feature_set_name,\n                    'reused': False\n                })\n                \n                completed += 1\n                print(f\" → CV: {cv_score:.4f} ✅\")\n                \n                # Show progress\n                elapsed = (datetime.now() - start_time).total_seconds()\n                if completed > 0 and elapsed > 0:\n                    rate = completed / elapsed\n                    remaining = (len(all_configs) - idx) / rate if rate > 0 else 0\n                    print(f\"  ⏱️ Progress: {elapsed/60:.1f}min elapsed, ~{remaining/60:.1f}min remaining\")\n            else:\n                print(\" → Failed ❌\")\n    else:\n        print(\"\\n✨ All configurations already computed!\")\n    \n    # Create ensembles (include reused predictions if available)\n    print(f\"\\n🎯 Total models for ensemble: {len(model_results)}\")\n    if len(model_results) >= 3:\n        ensemble_files = create_fast_ensembles(predictions_cache, submission_df, model_results, score_tracker)\n        submissions_created.extend(ensemble_files)\n    \n    # Save run summary\n    run_summary = {\n        \"models_trained\": completed,\n        \"models_reused\": skipped_count,\n        \"total_models\": len(model_results),\n        \"submissions_created\": len(submissions_created),\n        \"feature_sets_tested\": len(feature_sets_to_test),\n        \"best_cv\": max([m['cv_score'] for m in model_results]) if model_results else 0,\n        \"duration_minutes\": (datetime.now() - start_time).total_seconds() / 60\n    }\n    score_tracker.add_run_summary(run_summary)\n    \n    print(f\"\\n✅ Pipeline Complete!\")\n    print(f\"  - New models trained: {completed}\")\n    print(f\"  - Previous scores reused: {skipped_count}\")\n    print(f\"  - Feature sets tested: {len(feature_sets_to_test)}\")\n    print(f\"  - Submissions created: {len(submissions_created)}\")\n    print(f\"  - Total time: {run_summary['duration_minutes']:.1f} minutes\")\n    \n    # Show all-time best scores\n    print(\"\\n🏆 ALL-TIME BEST SCORES:\")\n    best_all_time = score_tracker.get_best_configs(10)\n    for i, (config_hash, score_data) in enumerate(best_all_time, 1):\n        model_info = score_data.get('model_info', {})\n        if isinstance(model_info, dict):\n            model_type = model_info.get('model', model_info.get('type', 'Unknown'))\n            strategy = model_info.get('strategy', model_info.get('method', ''))\n            feature_set = model_info.get('feature_set', 'default')\n        else:\n            model_type = 'Unknown'\n            strategy = ''\n            feature_set = 'default'\n        print(f\"  {i:2}. {model_type:10} {strategy:15} {feature_set:20} CV: {score_data['cv_score']:.4f}\")\n    \n    # Show best scores by feature set\n    print(\"\\n🏆 BEST SCORES BY FEATURE SET:\")\n    best_by_fs = score_tracker.get_best_by_feature_set(n=2)\n    for fs, models in best_by_fs.items():\n        if models:\n            print(f\"\\n{fs}:\")\n            for _, score_data in models:\n                model_info = score_data.get('model_info', {})\n                model_type = model_info.get('model', 'Unknown') if isinstance(model_info, dict) else 'Unknown'\n                strategy = model_info.get('strategy', '') if isinstance(model_info, dict) else ''\n                print(f\"  {model_type:10} {strategy:15} CV: {score_data['cv_score']:.4f}\")\n    \n    # Show feature set performance summary\n    score_tracker.print_feature_set_summary()\n    \n    # Export scores for next run\n    export_file = score_tracker.export_to_csv(\"scores_export.csv\")\n    if export_file:\n        print(f\"\\n💾 Scores saved to: {score_tracker.scores_file}\")\n        print(f\"📥 CSV export: {export_file}\")\n    \n    print(\"\\n📝 NEXT STEPS:\")\n    print(\"1. Submit the top files to Kaggle\")\n    print(\"2. Download 'scores_export.csv' and 'crypto_scores_history.json'\")\n    print(\"3. Create a new dataset with these files\")\n    print(\"4. Update FastConfig.UPLOADED_SCORES_PATHS with the new dataset path\")\n    print(\"5. Run again to continue building on your results!\")\n    \n    return submissions_created\n\n# ===== Main Execution =====\nif __name__ == \"__main__\":\n    print(\"🚀 FAST CRYPTO PIPELINE v6.0\")\n    print(\"✅ Advanced linear models (ElasticNet, Lasso, Huber, SVR)\")\n    print(\"✅ Feature combinations for X features\")\n    print(\"✅ Granular temporal strategies (90-99% recent data)\")\n    print(\"✅ Improved score tracking and error handling\")\n    print(\"=\" * 80)\n    \n    # Configure your uploaded score file paths here\n    uploaded_score_paths = [\n        # '/kaggle/input/1-set-of-test-records-drw/crypto_scores_history.json',\n        # Add more paths as needed:\n        # '/kaggle/input/crypto-scores-run2/scores_export.csv',\n        # '/kaggle/input/crypto-scores-run3/crypto_scores_history.json',\n    ]\n    \n    # Check which files exist\n    existing_files = []\n    for path in uploaded_score_paths:\n        if os.path.exists(path):\n            existing_files.append(path)\n            print(f\"✅ Found: {os.path.basename(path)}\")\n        else:\n            print(f\"❌ Not found: {os.path.basename(path)}\")\n    \n    # Choose which feature sets to test (None = test all)\n    # You can limit to specific sets like: [\"top_features\", \"extended_v1\", \"x_features_combinations\"]\n    feature_sets_to_test = None  # Will test all defined feature sets\n    \n    # Or test only specific feature sets:\n    # feature_sets_to_test = [\"top_features\", \"x_features_only\", \"x_features_combinations\"]\n    \n    # Run pipeline\n    submissions = run_fast_pipeline(\n        uploaded_scores_files=existing_files,\n        feature_sets_to_test=feature_sets_to_test\n    )\n    \n    print(\"\\n🎉 Pipeline completed successfully!\")\n    print(f\"Created {len(submissions)} submission files\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}