{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\nimport time\nimport warnings\nimport gc\nimport os\nfrom datetime import datetime\n\nwarnings.filterwarnings('ignore')\n\n\nclass CryptoMarketPredictor:\n    \"\"\"\n    A comprehensive framework for training crypto market prediction models\n    using different feature selection strategies and generating competition submissions.\n    \"\"\"\n    \n    def __init__(self, train_path, test_path, sulov_results_path='sulov_selection_results.json'):\n        self.train_path = train_path\n        self.test_path = test_path\n        self.sulov_results_path = sulov_results_path\n        self.baseline_features = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n        self.models = {}\n        self.feature_sets = {}\n        self.predictions = {}\n        self.performance_metrics = {}\n        \n    def load_sulov_results(self):\n        \"\"\"Load and parse SULOV feature selection results from JSON file.\"\"\"\n        print(\"Loading SULOV feature selection results...\")\n        \n        try:\n            with open(self.sulov_results_path, 'r') as f:\n                sulov_results = json.load(f)\n        except FileNotFoundError:\n            print(f\"Warning: {self.sulov_results_path} not found. Using default features.\")\n            return [], []\n        \n        selected_features = sulov_results.get('selected_features', [])\n        cluster_summary = sulov_results.get('cluster_summary', {})\n        \n        # Extract features with their target correlations\n        feature_correlations = []\n        for cluster_info in cluster_summary.values():\n            feature = cluster_info['representative']\n            correlation = cluster_info['target_correlation']\n            if feature not in self.baseline_features:\n                feature_correlations.append((feature, correlation))\n        \n        # Sort by correlation and take top 30\n        feature_correlations.sort(key=lambda x: x[1], reverse=True)\n        top_30_sulov = [feature for feature, _ in feature_correlations[:30]]\n        \n        print(f\"Loaded {len(selected_features)} total SULOV features\")\n        print(f\"Selected top 30 features by target correlation\")\n        \n        return top_30_sulov, feature_correlations\n    \n    def select_random_features(self, all_features, n_features=30, seed=42):\n        \"\"\"Select random features from the anonymized feature set.\"\"\"\n        np.random.seed(seed)\n        \n        # Filter out baseline features and any non-feature columns\n        anonymized_features = [f for f in all_features \n                             if f not in self.baseline_features \n                             and f not in ['timestamp', 'label']]\n        \n        # Ensure we have enough features to select from\n        if len(anonymized_features) < n_features:\n            print(f\"Warning: Only {len(anonymized_features)} features available, selecting all\")\n            return anonymized_features\n        \n        # Randomly select features\n        random_features = np.random.choice(anonymized_features, size=n_features, replace=False)\n        \n        print(f\"Selected {n_features} random features from {len(anonymized_features)} available\")\n        \n        return list(random_features)\n    \n    def load_and_prepare_data(self, feature_list, sample_size=None):\n        \"\"\"\n        Load training and test data with specified features.\n        \n        Parameters:\n        -----------\n        feature_list : list\n            List of features to use\n        sample_size : int or None\n            If specified, sample this many rows from training data\n        \n        Returns:\n        --------\n        tuple\n            (X_train, y_train, X_test, test_ids)\n        \"\"\"\n        print(f\"\\nLoading data with {len(feature_list)} features...\")\n        \n        # Load training data\n        print(\"Loading training data...\")\n        train_df = pd.read_parquet(self.train_path)\n        \n        # Apply sampling if requested\n        if sample_size and sample_size < len(train_df):\n            n_rows = len(train_df)\n            \n            # Sample more heavily from recent data\n            weights = np.linspace(0.5, 1.0, n_rows)\n            weights = weights / weights.sum()\n            \n            sample_indices = np.random.choice(n_rows, size=sample_size, replace=False, p=weights)\n            train_df = train_df.iloc[sample_indices].reset_index(drop=True)\n            print(f\"Sampled {sample_size} rows from {n_rows} total rows\")\n        \n        # Load test data\n        print(\"Loading test data...\")\n        test_df = pd.read_parquet(self.test_path)\n        \n        # Verify all requested features exist\n        available_features = [f for f in feature_list if f in train_df.columns]\n        missing_features = [f for f in feature_list if f not in train_df.columns]\n        \n        if missing_features:\n            print(f\"Warning: {len(missing_features)} features not found in data: {missing_features[:5]}...\")\n            feature_list = available_features\n        \n        # Extract features and target\n        X_train = train_df[feature_list].copy()\n        y_train = train_df['label'].copy()\n        X_test = test_df[feature_list].copy()\n        test_ids = test_df.index\n        \n        print(f\"Training data shape: {X_train.shape}\")\n        print(f\"Test data shape: {X_test.shape}\")\n        \n        # Handle any remaining missing values\n        for col in X_train.columns:\n            if X_train[col].isna().any():\n                median_val = X_train[col].median()\n                X_train[col] = X_train[col].fillna(median_val)\n                X_test[col] = X_test[col].fillna(median_val)\n        \n        # Clean up memory\n        del train_df, test_df\n        gc.collect()\n        \n        return X_train, y_train, X_test, test_ids\n    \n    def create_model_configurations(self):\n        \"\"\"Define model configurations for different approaches.\"\"\"\n        return {\n            'xgboost_conservative': {\n                'model_class': XGBRegressor,\n                'params': {\n                    'n_estimators': 300,\n                    'max_depth': 5,\n                    'learning_rate': 0.01,\n                    'subsample': 0.8,\n                    'colsample_bytree': 0.8,\n                    'gamma': 1,\n                    'reg_alpha': 0.1,\n                    'reg_lambda': 1,\n                    'random_state': 42,\n                    'n_jobs': -1,\n                    'tree_method': 'hist',\n                    'objective': 'reg:squarederror'\n                }\n            },\n            'xgboost_aggressive': {\n                'model_class': XGBRegressor,\n                'params': {\n                    'n_estimators': 500,\n                    'max_depth': 8,\n                    'learning_rate': 0.03,\n                    'subsample': 0.7,\n                    'colsample_bytree': 0.7,\n                    'gamma': 0.1,\n                    'reg_alpha': 0.01,\n                    'reg_lambda': 0.1,\n                    'random_state': 42,\n                    'n_jobs': -1,\n                    'tree_method': 'hist',\n                    'objective': 'reg:squarederror'\n                }\n            },\n            'lightgbm': {\n                'model_class': LGBMRegressor,\n                'params': {\n                    'n_estimators': 400,\n                    'num_leaves': 31,\n                    'learning_rate': 0.02,\n                    'feature_fraction': 0.8,\n                    'bagging_fraction': 0.8,\n                    'bagging_freq': 5,\n                    'reg_alpha': 0.1,\n                    'reg_lambda': 0.1,\n                    'min_child_samples': 20,\n                    'random_state': 42,\n                    'n_jobs': -1,\n                    'verbose': -1,\n                    'metric': 'rmse'\n                }\n            }\n        }\n    \n    def train_model_with_cv(self, X_train, y_train, model_config, n_folds=5):\n        \"\"\"\n        Train a model using cross-validation and return predictions and performance metrics.\n        \n        Parameters:\n        -----------\n        X_train : pandas.DataFrame\n            Training features\n        y_train : pandas.Series\n            Training target\n        model_config : dict\n            Model configuration including class and parameters\n        n_folds : int\n            Number of cross-validation folds\n        \n        Returns:\n        --------\n        tuple\n            (trained_model, cv_scores, feature_importance)\n        \"\"\"\n        model_name = model_config['model_class'].__name__\n        print(f\"\\nTraining {model_name} with {n_folds}-fold CV...\")\n        \n        # Initialize model\n        model_class = model_config['model_class']\n        params = model_config['params']\n        \n        # Prepare for cross-validation\n        kf = KFold(n_splits=n_folds, shuffle=True, random_state=42)\n        cv_scores = []\n        feature_importance = np.zeros(len(X_train.columns))\n        \n        # Standardize features\n        scaler = StandardScaler()\n        X_scaled = scaler.fit_transform(X_train)\n        \n        # Cross-validation\n        for fold, (train_idx, val_idx) in enumerate(kf.split(X_scaled)):\n            X_fold_train = X_scaled[train_idx]\n            y_fold_train = y_train.iloc[train_idx]\n            X_fold_val = X_scaled[val_idx]\n            y_fold_val = y_train.iloc[val_idx]\n            \n            # Train model\n            model = model_class(**params)\n            \n            # Fit with early stopping for tree-based models\n            if model_name in ['XGBRegressor', 'LGBMRegressor']:\n                model.fit(\n                    X_fold_train, \n                    y_fold_train,\n                    eval_set=[(X_fold_val, y_fold_val)],\n                    eval_metric='rmse',\n                    early_stopping_rounds=50,\n                    verbose=False\n                )\n            else:\n                model.fit(X_fold_train, y_fold_train)\n            \n            # Validate\n            val_predictions = model.predict(X_fold_val)\n            \n            # Calculate correlation\n            try:\n                correlation = pearsonr(y_fold_val, val_predictions)[0]\n                if np.isnan(correlation):\n                    correlation = 0.0\n            except:\n                correlation = 0.0\n                \n            cv_scores.append(correlation)\n            \n            # Accumulate feature importance\n            if hasattr(model, 'feature_importances_'):\n                feature_importance += model.feature_importances_ / n_folds\n            \n            print(f\"  Fold {fold + 1}: Correlation = {correlation:.6f}\")\n        \n        # Train final model on full data\n        print(\"  Training final model on full dataset...\")\n        final_model = model_class(**params)\n        \n        if model_name in ['XGBRegressor', 'LGBMRegressor']:\n            # Use a portion for validation\n            val_size = int(0.1 * len(X_scaled))\n            X_train_final = X_scaled[:-val_size]\n            y_train_final = y_train.iloc[:-val_size]\n            X_val_final = X_scaled[-val_size:]\n            y_val_final = y_train.iloc[-val_size:]\n            \n            final_model.fit(\n                X_train_final, \n                y_train_final,\n                eval_set=[(X_val_final, y_val_final)],\n                eval_metric='rmse',\n                early_stopping_rounds=50,\n                verbose=False\n            )\n        else:\n            final_model.fit(X_scaled, y_train)\n        \n        # Store scaler with model for later use\n        final_model.scaler = scaler\n        final_model.feature_names = list(X_train.columns)\n        \n        mean_score = np.mean(cv_scores)\n        std_score = np.std(cv_scores)\n        print(f\"  Mean CV Correlation: {mean_score:.6f} (±{std_score:.6f})\")\n        \n        return final_model, cv_scores, feature_importance\n    \n    def generate_predictions(self, model, X_test):\n        \"\"\"Generate predictions for test data using a trained model.\"\"\"\n        # Apply the same scaling used during training\n        X_test_scaled = model.scaler.transform(X_test)\n        predictions = model.predict(X_test_scaled)\n        \n        return predictions\n    \n    def create_ensemble_predictions(self, predictions_dict, weights=None):\n        \"\"\"\n        Create ensemble predictions from multiple models.\n        \n        Parameters:\n        -----------\n        predictions_dict : dict\n            Dictionary of model predictions\n        weights : dict or None\n            Optional weights for each model (defaults to equal weights)\n        \n        Returns:\n        --------\n        numpy.ndarray\n            Ensemble predictions\n        \"\"\"\n        if weights is None:\n            weights = {name: 1.0 / len(predictions_dict) for name in predictions_dict}\n        \n        # Ensure weights sum to 1\n        total_weight = sum(weights.values())\n        weights = {k: v/total_weight for k, v in weights.items()}\n        \n        ensemble_pred = np.zeros_like(next(iter(predictions_dict.values())))\n        \n        for name, pred in predictions_dict.items():\n            ensemble_pred += weights.get(name, 0) * pred\n        \n        return ensemble_pred\n    \n    def save_submission(self, predictions, test_ids, filename):\n        \"\"\"Save predictions in competition submission format.\"\"\"\n        submission = pd.DataFrame({\n            'id': test_ids,\n            'prediction': predictions\n        })\n        \n        submission.to_csv(filename, index=False)\n        print(f\"Saved submission to {filename}\")\n        \n        # Display sample predictions\n        print(f\"  Sample predictions: {predictions[:5]}\")\n        print(f\"  Prediction statistics: mean={np.mean(predictions):.6f}, std={np.std(predictions):.6f}\")\n        \n        return submission\n    \n    def run_complete_pipeline(self, use_sample=False, sample_size=100000):\n        \"\"\"\n        Execute the complete training and submission generation pipeline.\n        \n        Parameters:\n        -----------\n        use_sample : bool\n            Whether to use a sample of the data for faster iteration\n        sample_size : int\n            Number of rows to sample if use_sample is True\n        \"\"\"\n        print(\"=\"*80)\n        print(\"CRYPTO MARKET PREDICTION MODEL TRAINING\")\n        print(\"=\"*80)\n        \n        timestamp = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n        \n        # Step 1: Load SULOV results and prepare feature sets\n        print(\"\\nStep 1: Preparing feature sets...\")\n        \n        # First, load a small sample to get all feature names\n        print(\"Reading feature names from training data...\")\n        train_sample = pd.read_parquet(self.train_path)\n        all_features = [col for col in train_sample.columns if col not in ['timestamp', 'label']]\n        print(f\"Total features available: {len(all_features)}\")\n        \n        # Clean up sample\n        del train_sample\n        gc.collect()\n        \n        # Get SULOV features\n        top_30_sulov, feature_correlations = self.load_sulov_results()\n        \n        # If no SULOV features found, use a default set\n        if not top_30_sulov:\n            print(\"Warning: No SULOV features found, using first 30 anonymized features\")\n            anonymized = [f for f in all_features if f.startswith('X')]\n            top_30_sulov = anonymized[:30]\n        \n        sulov_features = self.baseline_features + top_30_sulov\n        \n        # Get random features\n        random_30 = self.select_random_features(all_features)\n        random_features = self.baseline_features + random_30\n        \n        # Store feature sets\n        self.feature_sets = {\n            'sulov': sulov_features,\n            'random': random_features\n        }\n        \n        # Print feature set summaries\n        print(f\"\\nSULOV feature set: {len(sulov_features)} features\")\n        print(f\"  Baseline: {self.baseline_features}\")\n        print(f\"  Top 5 SULOV: {top_30_sulov[:5]}\")\n        \n        print(f\"\\nRandom feature set: {len(random_features)} features\")\n        print(f\"  Baseline: {self.baseline_features}\")\n        print(f\"  First 5 random: {random_30[:5]}\")\n        \n        # Step 2: Load data for each feature set and train models\n        print(\"\\nStep 2: Loading data and training models...\")\n        \n        model_configs = self.create_model_configurations()\n        \n        # Determine sample size for training\n        train_sample_size = sample_size if use_sample else None\n        \n        for feature_set_name, features in self.feature_sets.items():\n            print(f\"\\n{'='*60}\")\n            print(f\"Training models with {feature_set_name.upper()} features\")\n            print(f\"{'='*60}\")\n            \n            # Load data\n            X_train, y_train, X_test, test_ids = self.load_and_prepare_data(\n                features, \n                sample_size=train_sample_size\n            )\n            \n            # Train different model configurations\n            set_predictions = {}\n            \n            for model_name, model_config in model_configs.items():\n                model_key = f\"{feature_set_name}_{model_name}\"\n                \n                try:\n                    # Train model\n                    model, cv_scores, feature_importance = self.train_model_with_cv(\n                        X_train, y_train, model_config\n                    )\n                    \n                    # Generate predictions\n                    predictions = self.generate_predictions(model, X_test)\n                    \n                    # Store results\n                    self.models[model_key] = model\n                    self.predictions[model_key] = predictions\n                    self.performance_metrics[model_key] = {\n                        'cv_scores': cv_scores,\n                        'mean_cv': np.mean(cv_scores),\n                        'std_cv': np.std(cv_scores),\n                        'feature_importance': feature_importance\n                    }\n                    \n                    set_predictions[model_name] = predictions\n                    \n                except Exception as e:\n                    print(f\"  Error training {model_key}: {str(e)}\")\n                    continue\n            \n            # Create ensemble for this feature set\n            if set_predictions:\n                ensemble_key = f\"{feature_set_name}_ensemble\"\n                ensemble_predictions = self.create_ensemble_predictions(set_predictions)\n                self.predictions[ensemble_key] = ensemble_predictions\n                print(f\"\\nCreated ensemble for {feature_set_name} with {len(set_predictions)} models\")\n        \n        # Step 3: Generate submissions\n        print(\"\\n\" + \"=\"*80)\n        print(\"Step 3: Generating submission files...\")\n        print(\"=\"*80)\n        \n        # Ensure we have test_ids\n        if 'test_ids' not in locals():\n            _, _, _, test_ids = self.load_and_prepare_data(self.baseline_features)\n        \n        # Submission 1: Best SULOV model (highest CV score)\n        sulov_models = {k: v for k, v in self.performance_metrics.items() if k.startswith('sulov_')}\n        \n        if sulov_models:\n            best_sulov_model = max(sulov_models.keys(), key=lambda x: sulov_models[x]['mean_cv'])\n            print(f\"\\nBest SULOV model: {best_sulov_model}\")\n            print(f\"  CV Score: {sulov_models[best_sulov_model]['mean_cv']:.6f}\")\n            \n            submission_1 = self.save_submission(\n                self.predictions[best_sulov_model],\n                test_ids,\n                f\"submission_1_best_sulov_{timestamp}.csv\"\n            )\n        \n        # Submission 2: SULOV ensemble\n        if 'sulov_ensemble' in self.predictions:\n            print(\"\\nGenerating SULOV ensemble submission...\")\n            submission_2 = self.save_submission(\n                self.predictions['sulov_ensemble'],\n                test_ids,\n                f\"submission_2_sulov_ensemble_{timestamp}.csv\"\n            )\n        \n        # Submission 3: Mixed ensemble (SULOV + Random)\n        if 'sulov_ensemble' in self.predictions and 'random_ensemble' in self.predictions:\n            print(\"\\nGenerating mixed ensemble submission...\")\n            mixed_ensemble = self.create_ensemble_predictions({\n                'sulov': self.predictions['sulov_ensemble'],\n                'random': self.predictions['random_ensemble']\n            }, weights={'sulov': 0.7, 'random': 0.3})\n            \n            submission_3 = self.save_submission(\n                mixed_ensemble,\n                test_ids,\n                f\"submission_3_mixed_ensemble_{timestamp}.csv\"\n            )\n        \n        # Step 4: Generate performance report\n        self.generate_performance_report(timestamp)\n        \n        print(\"\\n\" + \"=\"*80)\n        print(\"PIPELINE COMPLETED SUCCESSFULLY\")\n        print(\"=\"*80)\n        \n        return True\n    \n    def generate_performance_report(self, timestamp):\n        \"\"\"Generate a comprehensive performance report.\"\"\"\n        report_path = f\"model_performance_report_{timestamp}.txt\"\n        \n        with open(report_path, 'w') as f:\n            f.write(\"CRYPTO MARKET PREDICTION MODEL PERFORMANCE REPORT\\n\")\n            f.write(\"=\"*80 + \"\\n\")\n            f.write(f\"Generated: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\\n\\n\")\n            \n            # Feature set comparison\n            f.write(\"FEATURE SET COMPARISON\\n\")\n            f.write(\"-\"*40 + \"\\n\")\n            \n            # Calculate average performance by feature set\n            for feature_set in ['sulov', 'random']:\n                set_metrics = [(k, v) for k, v in self.performance_metrics.items() \n                             if k.startswith(f\"{feature_set}_\")]\n                \n                if set_metrics:\n                    avg_cv = np.mean([m[1]['mean_cv'] for m in set_metrics])\n                    \n                    f.write(f\"\\n{feature_set.upper()} Features:\\n\")\n                    f.write(f\"  Number of features: {len(self.feature_sets.get(feature_set, []))}\\n\")\n                    f.write(f\"  Average CV correlation: {avg_cv:.6f}\\n\")\n                    \n                    if feature_set in self.feature_sets:\n                        feature_preview = self.feature_sets[feature_set][:10]\n                        f.write(f\"  First 10 features: {', '.join(feature_preview)}\\n\")\n            \n            # Individual model performance\n            f.write(\"\\n\\nINDIVIDUAL MODEL PERFORMANCE\\n\")\n            f.write(\"-\"*40 + \"\\n\")\n            \n            if self.performance_metrics:\n                sorted_models = sorted(self.performance_metrics.items(), \n                                     key=lambda x: x[1]['mean_cv'], reverse=True)\n                \n                for model_name, metrics in sorted_models:\n                    f.write(f\"\\n{model_name}:\\n\")\n                    f.write(f\"  Mean CV Correlation: {metrics['mean_cv']:.6f}\\n\")\n                    f.write(f\"  Std CV Correlation: {metrics['std_cv']:.6f}\\n\")\n                    f.write(f\"  CV Scores: {[f'{s:.6f}' for s in metrics['cv_scores']]}\\n\")\n            \n            # Feature importance for top models\n            f.write(\"\\n\\nTOP FEATURE IMPORTANCE\\n\")\n            f.write(\"-\"*40 + \"\\n\")\n            \n            # Get best model from each feature set\n            for feature_set in ['sulov', 'random']:\n                set_models = {k: v for k, v in self.performance_metrics.items() \n                            if k.startswith(f\"{feature_set}_\")}\n                \n                if set_models:\n                    best_model_key = max(set_models.keys(), \n                                       key=lambda x: set_models[x]['mean_cv'])\n                    \n                    if best_model_key in self.models:\n                        model = self.models[best_model_key]\n                        importance = self.performance_metrics[best_model_key].get('feature_importance', [])\n                        \n                        if len(importance) > 0 and hasattr(model, 'feature_names'):\n                            f.write(f\"\\n{best_model_key} - Top 10 features:\\n\")\n                            \n                            # Get feature importance with names\n                            feature_imp = list(zip(model.feature_names, importance))\n                            feature_imp.sort(key=lambda x: x[1], reverse=True)\n                            \n                            for i, (feat, imp) in enumerate(feature_imp[:10]):\n                                f.write(f\"  {i+1}. {feat}: {imp:.4f}\\n\")\n            \n            # Submission descriptions\n            f.write(\"\\n\\nSUBMISSION DESCRIPTIONS\\n\")\n            f.write(\"-\"*40 + \"\\n\")\n            f.write(\"\\nSubmission 1: Best performing SULOV-based model\\n\")\n            f.write(\"  - Uses features selected through SULOV clustering algorithm\\n\")\n            f.write(\"  - Single model with highest cross-validation performance\\n\")\n            \n            f.write(\"\\nSubmission 2: Ensemble of all SULOV-based models\\n\")\n            f.write(\"  - Combines predictions from XGBoost and LightGBM variants\\n\")\n            f.write(\"  - Equal weighting of all models\\n\")\n            \n            f.write(\"\\nSubmission 3: Mixed ensemble (70% SULOV, 30% Random)\\n\")\n            f.write(\"  - Weighted combination of SULOV and random feature models\\n\")\n            f.write(\"  - Provides robustness against feature selection bias\\n\")\n        \n        print(f\"\\nPerformance report saved to {report_path}\")\n\n\ndef main():\n    \"\"\"Execute the complete model training and submission generation pipeline.\"\"\"\n    # Initialize predictor\n    predictor = CryptoMarketPredictor(\n        train_path=\"/kaggle/input/drw-crypto-market-prediction/train.parquet\",\n        test_path=\"/kaggle/input/drw-crypto-market-prediction/test.parquet\",\n        sulov_results_path=\"sulov_selection_results.json\"\n    )\n    \n    # Run complete pipeline\n    # Set use_sample=True and adjust sample_size for faster testing\n    success = predictor.run_complete_pipeline(use_sample=False)\n    \n    if success:\n        # Display final summary\n        print(\"\\nFinal Summary:\")\n        print(\"-\"*40)\n        print(\"Three submission files have been generated:\")\n        print(\"1. Best SULOV model - Single model with highest CV performance\")\n        print(\"2. SULOV ensemble - Combination of all SULOV-based models\")\n        print(\"3. Mixed ensemble - Weighted combination of SULOV and random features\")\n        print(\"\\nReview the performance report for detailed metrics and comparisons.\")\n        \n        # Display model performance summary\n        if predictor.performance_metrics:\n            print(\"\\nModel Performance Summary:\")\n            print(\"-\"*40)\n            \n            # Calculate average performance by feature set\n            for feature_set in ['sulov', 'random']:\n                models = [(k, v['mean_cv']) for k, v in predictor.performance_metrics.items() \n                         if k.startswith(f\"{feature_set}_\")]\n                \n                if models:\n                    avg_score = np.mean([score for _, score in models])\n                    best_model = max(models, key=lambda x: x[1])\n                    \n                    print(f\"\\n{feature_set.upper()} Features:\")\n                    print(f\"  Average CV Score: {avg_score:.6f}\")\n                    print(f\"  Best Model: {best_model[0]} (Score: {best_model[1]:.6f})\")\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}