{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12566011,"sourceType":"datasetVersion","datasetId":7935367}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":679.589845,"end_time":"2025-07-04T10:52:52.29918","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-07-04T10:41:32.709335","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold, TimeSeriesSplit\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\nimport optuna\nfrom optuna.samplers import TPESampler\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# ==================== Feature Engineering ==================== #\n\ndef feature_engineering(df):\n    \"\"\"Apply comprehensive feature engineering to the dataframe\"\"\"\n    # Exponential combinations\n    df['exp_856P868P855P289'] = np.exp(df['X446'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_860P868P855P289'] = np.exp(df['X3'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_598P868P855P289'] = np.exp(df['X66'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_612P868P855P289'] = np.exp(df['X1'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_289P855P21'] = np.exp(df['X287'] + df['X37'] + df['X21'])\n    df['868xexp_289M125'] = df['X76'] * np.exp(df['X287'] - df['X19'])\n    \n    df['exp_603P868P855P289'] = np.exp(df['X25'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_174P868P855P289'] = np.exp(df['X174'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_465P868P855P289'] = np.exp(df['X465'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_125P862P289M125'] = np.exp(df['X19'] + df['X123'] + df['X287'] - df['X19'])\n    df['exp_168P868P855P289'] = np.exp(df['X168'] + df['X76'] + df['X37'] + df['X287'])\n    df['exp_855P289M125'] = np.exp(df['X37'] + df['X287'] - df['X19'])\n    df['exp_302P289M125'] = np.exp(df['X298'] + df['X287'] - df['X19'])\n    df['289xexp_289M125'] = df['X287'] * np.exp(df['X287'] - df['X19'])\n    \n    df['exp_862P868P855P289'] = np.exp(df['X123'] + df['X76'] + df['X37'] + df['X287'])\n    df['868x868x855x289'] = df['X76'] * df['X76'] * df['X37'] * df['X287']\n    df['385xexp_289M125'] = df['X385'] * np.exp(df['X287'] - df['X19'])\n    df['exp_862P289M125'] = np.exp(df['X123'] + df['X287'] - df['X19'])\n    df['exp_786P289M125'] = np.exp(df['X453'] + df['X287'] - df['X19'])\n    df['exp_856P289M125'] = np.exp(df['X446'] + df['X287'] - df['X19'])\n    df['852x868x855x289'] = df['X594'] * df['X76'] * df['X37'] * df['X287']\n    df['465x862x465'] = df['X465'] * df['X465'] * df['X123']\n    df['540x881'] = df['X476'] * df['X443']\n    \n    # Market microstructure features\n    df['bid_ask_interaction'] = df['ask_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    \n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n    \n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['ask_qty'] - df['ask_qty']) / (df['ask_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['ask_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    df['ask_buy_interaction_x_X293'] = df['X293'] * df['ask_buy_interaction']\n    \n    # Price pressure indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity depth measures\n    df['total_depth'] = df['ask_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['ask_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['ask_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order flow toxicity proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market activity indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['ask_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure volatility proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex interaction terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['ask_qty'] - df['ask_qty'])\n    \n    # Information asymmetry measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['ask_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market efficiency indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative measures\n    df['bid_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market stress indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['ask_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # Additional features\n    df['sqrt_volume_div_log_volume'] = df['sqrt_volume'] / (df['log_volume'] + 1e-6)\n    df['sqrt_volume_div_activity_intensity'] = df['sqrt_volume'] / (df['activity_intensity'] + 1e-6)\n    df['sqrt_volume_mul_fill_probability'] = df['sqrt_volume'] * df['fill_probability']\n    df['volume_div_sqrt_volume'] = df['volume'] / (df['sqrt_volume'] + 1e-6)\n    df['sqrt_volume_div_fill_probability'] = df['sqrt_volume'] / (df['fill_probability'] + 1e-6)\n    df['sqrt_volume_mul_activity_intensity'] = df['sqrt_volume'] * df['activity_intensity']\n    df['sqrt_volume_div_log_sell_qty'] = df['sqrt_volume'] / (df['log_sell_qty'] + 1e-6)\n    df['log_buy_qty_mul_sqrt_volume'] = df['log_buy_qty'] * df['sqrt_volume']\n    df['sqrt_volume_mul_log_buy_qty'] = df['sqrt_volume'] * df['log_buy_qty']\n    df['log_volume_mul_sqrt_volume'] = df['log_volume'] * df['sqrt_volume']\n    \n    df['log_sell_qty_mul_X598'] = df['log_sell_qty'] * df['X66']\n    df['log_buy_qty_mul_X598'] = df['log_buy_qty'] * df['X66']\n    df['log_volume_mul_X598'] = df['log_volume'] * df['X66']\n    df['sqrt_volume_mul_X856'] = df['sqrt_volume'] * df['X446']\n    df['log_sell_qty_mul_X302'] = df['log_sell_qty'] * df['X298']\n    df['log_volume_mul_X302'] = df['log_volume'] * df['X298']\n    df['log_buy_qty_mul_X302'] = df['log_buy_qty'] * df['X298']\n    df['log_sell_qty_mul_X292'] = df['log_sell_qty'] * df['X292']\n    \n    # Handle infinities and NaNs\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    return df\n\n# ==================== Configuration ==================== #\n\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    BASE_FEATURES = [\n        \"X287\", \"X446\", \"X66\", \"X123\", \"X385\", \"X594\", \"X25\", \"X3\", \"X231\",\n        \"X415\", \"X345\", \"X37\", \"X174\", \"X298\", \"X178\", \"X168\", \"X1\",\n        \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\", \"X210\", \"X421\", \"X92\",\n        'X465', 'X105', 'X19', 'X21', \"X76\", \"X453\", \"X293\", \"X504\", \n        'X476', 'X486', 'X443', 'X329', 'X79', \"X292\", \"X244\"\n    ]\n    \n    SELECTED_FEATURES = [\n        \"X287\", \"X446\", \"X66\", \"X123\", \"X385\", \"X25\", \"X3\", \"X231\",\n        \"X415\", \"X345\", \"X37\", \"X174\", \"X298\", \"X178\", \"X168\", \"X1\",\n        \"buy_qty\", \"sell_qty\", \"volume\", \n        \"X210\", \"X421\", \"X92\", \"X292\", \"X244\", \"X329\",\n        'ask_buy_interaction_x_X293', '868xexp_289M125', 'exp_786P289M125', 'exp_856P289M125',\n        'exp_612P868P855P289', 'exp_598P868P855P289', 'exp_855P289M125',\n        '385xexp_289M125', '465x862x465', '540x881', 'exp_125P862P289M125',\n        'bid_ask_interaction', 'bid_buy_interaction', 'bid_sell_interaction', \n        'ask_buy_interaction', 'ask_sell_interaction', \"log_volume\", 'net_order_flow', \n        'normalized_net_flow', 'buying_pressure', 'volume_weighted_buy', 'total_depth', \n        'depth_imbalance', 'relative_spread', 'log_depth', 'kyle_lambda', 'flow_toxicity', \n        'aggressive_flow_ratio', 'volume_depth_ratio', 'activity_intensity', 'log_buy_qty', \n        'log_sell_qty', 'log_bid_qty', 'log_ask_qty', 'realized_spread_proxy', \n        'price_impact_proxy', 'quote_volatility_proxy', 'flow_depth_interaction', \n        'imbalance_volume_interaction', 'depth_volume_interaction', 'trade_informativeness',\n        'execution_shortfall_proxy', 'adverse_selection_proxy', 'fill_probability',\n        'execution_rate', 'market_efficiency', 'sqrt_volume', 'sqrt_depth', 'volume_squared',\n        'imbalance_squared', 'bid_ratio', 'ask_ratio', 'buy_ratio', 'sell_ratio',\n        'liquidity_consumption', 'market_stress', 'depth_depletion', 'net_buying_ratio',\n        'directional_volume', 'signed_volume',\n        \"sqrt_volume_mul_fill_probability\", \"volume_div_sqrt_volume\",\n        \"log_buy_qty_mul_sqrt_volume\", \"sqrt_volume_mul_log_buy_qty\",\n        \"log_volume_mul_sqrt_volume\", \"sqrt_volume_mul_X856\",\n        \"log_sell_qty_mul_X302\", \"log_volume_mul_X302\",\n        \"log_buy_qty_mul_X302\", \"log_sell_qty_mul_X292\"\n    ]\n    \n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    RANDOM_STATE = 42\n    N_TRIALS = 100  # Number of Optuna trials\n    \n    # Stability parameters\n    MIN_SAMPLES_LEAF_RATIO = 0.01  # Minimum samples in leaf as ratio of training set\n    MAX_DEPTH_LIMIT = 15  # Maximum depth to prevent overfitting\n    STABILITY_WEIGHT = 0.3  # Weight for stability metric in optimization\n\n# ==================== Data Loading and Utilities ==================== #\n\ndef create_time_decay_weights(n: int, decay: float = 0.9) -> np.ndarray:\n    \"\"\"Create time-based sample weights with exponential decay\"\"\"\n    positions = np.arange(n)\n    normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()\n\ndef load_data():\n    \"\"\"Load and preprocess data\"\"\"\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=Config.BASE_FEATURES + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=Config.BASE_FEATURES)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n    \n    train_df = feature_engineering(train_df)\n    test_df = feature_engineering(test_df)\n    \n    print(f\"Data loaded - Train: {train_df.shape}, Test: {test_df.shape}\")\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\ndef calculate_stability_metrics(predictions_list):\n    \"\"\"Calculate stability metrics across multiple predictions\"\"\"\n    predictions_array = np.array(predictions_list)\n    \n    # Standard deviation across predictions\n    std_dev = np.std(predictions_array, axis=0)\n    mean_std = np.mean(std_dev)\n    \n    # Coefficient of variation\n    mean_pred = np.mean(predictions_array, axis=0)\n    cv = np.mean(std_dev / (np.abs(mean_pred) + 1e-10))\n    \n    # Inter-prediction correlation\n    correlations = []\n    for i in range(len(predictions_list)):\n        for j in range(i + 1, len(predictions_list)):\n            corr = pearsonr(predictions_list[i], predictions_list[j])[0]\n            correlations.append(corr)\n    \n    avg_correlation = np.mean(correlations) if correlations else 1.0\n    \n    return {\n        'mean_std': mean_std,\n        'cv': cv,\n        'avg_correlation': avg_correlation,\n        'stability_score': avg_correlation - cv  # Higher is better\n    }\n\n# ==================== Optuna Optimization ==================== #\n\ndef create_objective(train_df, config):\n    \"\"\"Create Optuna objective function with stability considerations\"\"\"\n    \n    def objective(trial):\n        # XGBoost hyperparameters with stability-aware constraints\n        params = {\n            'tree_method': 'hist',\n            'device': 'gpu',\n            'verbosity': 0,\n            'random_state': config.RANDOM_STATE,\n            'n_jobs': -1,\n            \n            # Tree-specific parameters\n            'n_estimators': trial.suggest_int('n_estimators', 100, 2000),\n            'max_depth': trial.suggest_int('max_depth', 3, config.MAX_DEPTH_LIMIT),\n            'min_child_weight': trial.suggest_int('min_child_weight', 10, 100),\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n            \n            # Regularization parameters (important for stability)\n            'reg_alpha': trial.suggest_float('reg_alpha', 0.1, 100, log=True),\n            'reg_lambda': trial.suggest_float('reg_lambda', 0.1, 100, log=True),\n            'gamma': trial.suggest_float('gamma', 0.1, 10, log=True),\n            \n            # Sampling parameters\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'colsample_bylevel': trial.suggest_float('colsample_bylevel', 0.5, 1.0),\n            'colsample_bynode': trial.suggest_float('colsample_bynode', 0.5, 1.0),\n        }\n        \n        # Use time series split for more realistic validation\n        tscv = TimeSeriesSplit(n_splits=config.N_FOLDS, test_size=len(train_df) // (config.N_FOLDS + 1))\n        \n        fold_scores = []\n        fold_predictions = []\n        stability_scores = []\n        \n        for fold, (train_idx, valid_idx) in enumerate(tscv.split(train_df)):\n            X_train = train_df.iloc[train_idx][config.SELECTED_FEATURES]\n            y_train = train_df.iloc[train_idx][config.LABEL_COLUMN]\n            X_valid = train_df.iloc[valid_idx][config.SELECTED_FEATURES]\n            y_valid = train_df.iloc[valid_idx][config.LABEL_COLUMN]\n            \n            # Apply time decay weights\n            sample_weights = create_time_decay_weights(len(train_idx))\n            \n            # Train model\n            model = XGBRegressor(**params)\n            model.fit(\n                X_train, y_train,\n                sample_weight=sample_weights,\n                eval_set=[(X_valid, y_valid)],\n                verbose=False\n            )\n            \n            # Predictions\n            valid_pred = model.predict(X_valid)\n            fold_predictions.append(valid_pred)\n            \n            # Performance metric\n            score = pearsonr(y_valid, valid_pred)[0]\n            fold_scores.append(score)\n            \n            # Additional stability check: predict on overlapping windows\n            if fold > 0:\n                # Check prediction consistency on previous validation set\n                prev_valid_idx = list(tscv.split(train_df))[fold-1][1]\n                overlap_idx = np.intersect1d(prev_valid_idx, valid_idx)\n                if len(overlap_idx) > 0:\n                    overlap_pred = model.predict(train_df.iloc[overlap_idx][config.SELECTED_FEATURES])\n                    stability_scores.append(np.std(overlap_pred))\n        \n        # Calculate overall metrics\n        avg_score = np.mean(fold_scores)\n        score_std = np.std(fold_scores)\n        \n        # Stability penalty\n        stability_penalty = score_std\n        if stability_scores:\n            stability_penalty += np.mean(stability_scores) * 0.1\n        \n        # Combined objective (maximize performance while minimizing instability)\n        objective_value = avg_score - config.STABILITY_WEIGHT * stability_penalty\n        \n        # Report intermediate values\n        trial.set_user_attr('avg_pearson', avg_score)\n        trial.set_user_attr('pearson_std', score_std)\n        trial.set_user_attr('stability_penalty', stability_penalty)\n        \n        return objective_value\n    \n    return objective\n\ndef optimize_hyperparameters(train_df, config, n_trials=None):\n    \"\"\"Run Optuna optimization with stability considerations\"\"\"\n    if n_trials is None:\n        n_trials = config.N_TRIALS\n    \n    # Create study with pruning for efficiency\n    study = optuna.create_study(\n        direction='maximize',\n        sampler=TPESampler(seed=config.RANDOM_STATE),\n        pruner=optuna.pruners.MedianPruner(n_startup_trials=10, n_warmup_steps=5)\n    )\n    \n    # Create and optimize objective\n    objective = create_objective(train_df, config)\n    \n    print(f\"Starting Optuna optimization with {n_trials} trials...\")\n    study.optimize(objective, n_trials=n_trials, show_progress_bar=True)\n    \n    # Print results\n    print(\"\\n=== Optimization Results ===\")\n    print(f\"Best objective value: {study.best_value:.4f}\")\n    print(f\"Best Pearson correlation: {study.best_trial.user_attrs['avg_pearson']:.4f}\")\n    print(f\"Pearson std deviation: {study.best_trial.user_attrs['pearson_std']:.4f}\")\n    print(f\"Stability penalty: {study.best_trial.user_attrs['stability_penalty']:.4f}\")\n    \n    print(\"\\nBest parameters:\")\n    for key, value in study.best_params.items():\n        print(f\"  {key}: {value}\")\n    \n    return study.best_params\n\n# ==================== Model Training with Stability ==================== #\n\ndef train_stable_ensemble(train_df, test_df, optimized_params, config):\n    \"\"\"Train ensemble with focus on stability\"\"\"\n    n_samples = len(train_df)\n    \n    # Define conservative data slices for stability\n    model_slices = [\n        {\"name\": \"full_data\", \"cutoff\": 0, \"weight\": 1.0},\n        {\"name\": \"recent_90pct\", \"cutoff\": int(0.10 * n_samples), \"weight\": 1.5},\n        {\"name\": \"recent_80pct\", \"cutoff\": int(0.20 * n_samples), \"weight\": 1.2},\n        {\"name\": \"recent_70pct\", \"cutoff\": int(0.30 * n_samples), \"weight\": 1.0},\n    ]\n    \n    # Initialize prediction storage\n    oof_preds = {s[\"name\"]: np.zeros(n_samples) for s in model_slices}\n    test_preds = {s[\"name\"]: np.zeros(len(test_df)) for s in model_slices}\n    \n    # Use both TimeSeriesSplit and KFold for robustness\n    tscv = TimeSeriesSplit(n_splits=config.N_FOLDS, test_size=n_samples // (config.N_FOLDS + 1))\n    kf = KFold(n_splits=config.N_FOLDS, shuffle=False)\n    \n    # Train models\n    for fold, (train_idx, valid_idx) in enumerate(tscv.split(train_df), 1):\n        print(f\"\\nTraining fold {fold}/{config.N_FOLDS}\")\n        \n        X_valid = train_df.iloc[valid_idx][config.SELECTED_FEATURES]\n        y_valid = train_df.iloc[valid_idx][config.LABEL_COLUMN]\n        \n        for slice_info in model_slices:\n            cutoff = slice_info[\"cutoff\"]\n            slice_name = slice_info[\"name\"]\n            \n            # Prepare training data\n            if cutoff > 0:\n                subset = train_df.iloc[cutoff:].reset_index(drop=True)\n                rel_idx = train_idx[train_idx >= cutoff] - cutoff\n            else:\n                subset = train_df\n                rel_idx = train_idx\n            \n            if len(rel_idx) == 0:\n                continue\n            \n            X_train = subset.iloc[rel_idx][config.SELECTED_FEATURES]\n            y_train = subset.iloc[rel_idx][config.LABEL_COLUMN]\n            \n            # Time decay weights\n            weights = create_time_decay_weights(len(subset))[rel_idx]\n            \n            # Train model with optimized parameters\n            model_params = optimized_params.copy()\n            model_params.update({\n                'tree_method': 'hist',\n                'device': 'gpu',\n                'verbosity': 0,\n                'random_state': config.RANDOM_STATE + fold,\n                'n_jobs': -1\n            })\n            \n            model = XGBRegressor(**model_params)\n            model.fit(\n                X_train, y_train,\n                sample_weight=weights,\n                eval_set=[(X_valid, y_valid)],\n                verbose=False\n            )\n            \n            # Generate predictions\n            valid_mask = valid_idx >= cutoff\n            if valid_mask.any():\n                valid_subset = valid_idx[valid_mask]\n                oof_preds[slice_name][valid_subset] = model.predict(\n                    train_df.iloc[valid_subset][config.SELECTED_FEATURES]\n                )\n            \n            # Test predictions\n            test_preds[slice_name] += model.predict(test_df[config.SELECTED_FEATURES])\n    \n    # Average test predictions\n    for slice_name in test_preds:\n        test_preds[slice_name] /= config.N_FOLDS\n    \n    return oof_preds, test_preds, model_slices\n\n# ==================== Final Ensemble and Submission ==================== #\n\ndef create_stable_submission(train_df, oof_preds, test_preds, model_slices, submission_df):\n    \"\"\"Create final predictions with stability weighting\"\"\"\n    \n    # Calculate stability metrics for each slice\n    slice_metrics = {}\n    for slice_info in model_slices:\n        slice_name = slice_info[\"name\"]\n        \n        # Performance metric\n        pearson = pearsonr(train_df[Config.LABEL_COLUMN], oof_preds[slice_name])[0]\n        \n        # Stability metric (lower variance is better)\n        pred_variance = np.var(oof_preds[slice_name])\n        \n        slice_metrics[slice_name] = {\n            'pearson': pearson,\n            'variance': pred_variance,\n            'base_weight': slice_info[\"weight\"]\n        }\n        \n        print(f\"{slice_name}: Pearson={pearson:.4f}, Variance={pred_variance:.4f}\")\n    \n    # Calculate adaptive weights based on performance and stability\n    weights = {}\n    for slice_name, metrics in slice_metrics.items():\n        # Combine performance and stability\n        performance_score = metrics['pearson']\n        stability_score = 1 / (1 + metrics['variance'])  # Lower variance = higher score\n        \n        # Final weight combines base weight, performance, and stability\n        weights[slice_name] = (\n            metrics['base_weight'] * \n            (0.7 * performance_score + 0.3 * stability_score)\n        )\n    \n    # Normalize weights\n    total_weight = sum(weights.values())\n    weights = {k: v / total_weight for k, v in weights.items()}\n    \n    print(\"\\nFinal ensemble weights:\")\n    for slice_name, weight in weights.items():\n        print(f\"  {slice_name}: {weight:.3f}\")\n    \n    # Create weighted ensemble\n    final_oof = sum(weights[s] * oof_preds[s] for s in weights)\n    final_test = sum(weights[s] * test_preds[s] for s in weights)\n    \n    # Final score\n    final_score = pearsonr(train_df[Config.LABEL_COLUMN], final_oof)[0]\n    print(f\"\\nFinal ensemble Pearson correlation: {final_score:.4f}\")\n    \n    # Create submission\n    submission_df[\"prediction\"] = final_test\n    submission_df.to_csv(\"submission_stable.csv\", index=False)\n    print(\"Submission saved to submission_stable.csv\")\n    \n    return submission_df, final_score\n\n# ==================== Main Execution ==================== #\n\ndef main():\n    \"\"\"Main execution function\"\"\"\n    # Load data\n    print(\"Loading data...\")\n    train_df, test_df, submission_df = load_data()\n    \n    # Optimize hyperparameters with stability focus\n    print(\"\\nOptimizing hyperparameters...\")\n    optimized_params = optimize_hyperparameters(\n        train_df, \n        Config, \n        n_trials=Config.N_TRIALS\n    )\n    \n    # Train stable ensemble\n    print(\"\\nTraining stable ensemble...\")\n    oof_preds, test_preds, model_slices = train_stable_ensemble(\n        train_df, \n        test_df, \n        optimized_params, \n        Config\n    )\n    \n    # Create final submission\n    print(\"\\nCreating final submission...\")\n    submission_df, final_score = create_stable_submission(\n        train_df, \n        oof_preds, \n        test_preds, \n        model_slices, \n        submission_df\n    )\n    \n    print(\"\\n=== Training Complete ===\")\n    print(f\"Final validation score: {final_score:.4f}\")\n    \n    return submission_df, optimized_params, final_score\n\nif __name__ == \"__main__\":\n    submission_df, best_params, score = main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}