{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.stats import pearsonr\nfrom scipy.optimize import minimize, differential_evolution\nfrom scipy.interpolate import interp1d\nimport optuna\nfrom typing import Dict, List, Tuple, Optional\nimport gc\n\nprint(\"Starting Intelligent XGBoost Optimization Pipeline...\")\nprint(\"=\" * 80)\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    n_folds = 5\n    random_state = 42\n    use_gpu = False\n\nclass IntelligentXGBOptimizer:\n    \"\"\"\n    Intelligent optimizer that uses gradient information and adaptive search\n    to find the optimal balance between training and CV performance\n    \"\"\"\n    \n    def __init__(self, X_train, y_train, X_test, n_folds=5):\n        self.X_train = X_train\n        self.y_train = y_train\n        self.X_test = X_test\n        self.n_folds = n_folds\n        self.performance_history = []\n        self.gradient_history = []\n        self.best_params = None\n        self.best_cv_score = -np.inf\n        \n    def compute_cv_gradient(self, params: Dict, epsilon: float = 1e-4) -> Dict:\n        \"\"\"\n        Compute gradient of CV score with respect to parameters\n        using finite differences\n        \"\"\"\n        base_score = self.evaluate_params(params)\n        gradients = {}\n        \n        for param_name, param_value in params.items():\n            if isinstance(param_value, (int, float)):\n                # Perturb parameter\n                params_plus = params.copy()\n                if isinstance(param_value, int):\n                    params_plus[param_name] = param_value + 1\n                else:\n                    params_plus[param_name] = param_value * (1 + epsilon)\n                \n                # Compute gradient\n                score_plus = self.evaluate_params(params_plus)\n                gradient = (score_plus - base_score) / (params_plus[param_name] - param_value)\n                gradients[param_name] = gradient\n        \n        return gradients\n    \n    def evaluate_params(self, params: Dict) -> float:\n        \"\"\"\n        Evaluate parameters using cross-validation\n        Returns negative of CV-train gap to encourage generalization\n        \"\"\"\n        kf = KFold(n_splits=self.n_folds, shuffle=True, random_state=42)\n        \n        cv_scores = []\n        train_scores = []\n        \n        for fold, (train_idx, valid_idx) in enumerate(kf.split(self.X_train)):\n            X_fold_train = self.X_train.iloc[train_idx]\n            y_fold_train = self.y_train.iloc[train_idx]\n            X_fold_valid = self.X_train.iloc[valid_idx]\n            y_fold_valid = self.y_train.iloc[valid_idx]\n            \n            # Train model\n            model = XGBRegressor(**params, random_state=42, n_jobs=-1, verbosity=0)\n            \n            # Use early stopping\n            model.fit(\n                X_fold_train, y_fold_train,\n                eval_set=[(X_fold_valid, y_fold_valid)],\n                early_stopping_rounds=50,\n                verbose=False\n            )\n            \n            # Evaluate\n            train_pred = model.predict(X_fold_train)\n            valid_pred = model.predict(X_fold_valid)\n            \n            train_score = pearsonr(y_fold_train, train_pred)[0]\n            valid_score = pearsonr(y_fold_valid, valid_pred)[0]\n            \n            train_scores.append(train_score)\n            cv_scores.append(valid_score)\n        \n        avg_train = np.mean(train_scores)\n        avg_cv = np.mean(cv_scores)\n        \n        # Penalize overfitting (large train-CV gap)\n        overfitting_penalty = max(0, avg_train - avg_cv - 0.05) ** 2\n        \n        # Combined score: maximize CV while minimizing overfitting\n        score = avg_cv - 0.5 * overfitting_penalty\n        \n        # Store history\n        self.performance_history.append({\n            'params': params.copy(),\n            'train_score': avg_train,\n            'cv_score': avg_cv,\n            'gap': avg_train - avg_cv,\n            'combined_score': score\n        })\n        \n        return score\n    \n    def adaptive_grid_search(self, param_space: Dict, n_iterations: int = 20):\n        \"\"\"\n        Adaptive grid search that focuses on promising regions\n        \"\"\"\n        print(\"\\nStarting adaptive grid search...\")\n        \n        # Initialize with random sampling\n        best_params = {}\n        for param, (min_val, max_val) in param_space.items():\n            if isinstance(min_val, int):\n                best_params[param] = np.random.randint(min_val, max_val + 1)\n            else:\n                best_params[param] = np.random.uniform(min_val, max_val)\n        \n        best_score = self.evaluate_params(best_params)\n        \n        # Adaptive search\n        for iteration in range(n_iterations):\n            print(f\"\\nIteration {iteration + 1}/{n_iterations}\")\n            \n            # Compute gradients\n            gradients = self.compute_cv_gradient(best_params)\n            self.gradient_history.append(gradients)\n            \n            # Update parameters using gradient ascent with adaptive learning rate\n            learning_rate = 0.1 / (1 + iteration * 0.1)\n            \n            new_params = best_params.copy()\n            for param, gradient in gradients.items():\n                if param in param_space:\n                    min_val, max_val = param_space[param]\n                    \n                    # Update with gradient\n                    if isinstance(best_params[param], int):\n                        update = int(learning_rate * gradient * (max_val - min_val))\n                        new_params[param] = np.clip(\n                            best_params[param] + update, min_val, max_val\n                        )\n                    else:\n                        update = learning_rate * gradient * (max_val - min_val)\n                        new_params[param] = np.clip(\n                            best_params[param] + update, min_val, max_val\n                        )\n            \n            # Evaluate new parameters\n            new_score = self.evaluate_params(new_params)\n            \n            # Update best if improved\n            if new_score > best_score:\n                best_score = new_score\n                best_params = new_params.copy()\n                print(f\"  Improved! Score: {best_score:.4f}\")\n            \n            # Add exploration noise occasionally\n            if iteration % 5 == 0:\n                for param in best_params:\n                    if param in param_space and np.random.random() < 0.3:\n                        min_val, max_val = param_space[param]\n                        noise_scale = 0.1 * (max_val - min_val)\n                        if isinstance(best_params[param], int):\n                            best_params[param] += int(np.random.normal(0, noise_scale))\n                            best_params[param] = np.clip(best_params[param], min_val, max_val)\n                        else:\n                            best_params[param] += np.random.normal(0, noise_scale)\n                            best_params[param] = np.clip(best_params[param], min_val, max_val)\n        \n        self.best_params = best_params\n        self.best_cv_score = best_score\n        return best_params\n    \n    def hill_climbing_optimizer(self, initial_params: Dict, param_space: Dict, \n                               n_restarts: int = 3, max_iterations: int = 50):\n        \"\"\"\n        Hill climbing with random restarts to escape local optima\n        \"\"\"\n        print(\"\\nStarting hill climbing optimization...\")\n        \n        global_best_params = initial_params.copy()\n        global_best_score = self.evaluate_params(initial_params)\n        \n        for restart in range(n_restarts):\n            print(f\"\\nRestart {restart + 1}/{n_restarts}\")\n            \n            # Random restart except first iteration\n            if restart > 0:\n                current_params = {}\n                for param, (min_val, max_val) in param_space.items():\n                    if isinstance(min_val, int):\n                        current_params[param] = np.random.randint(min_val, max_val + 1)\n                    else:\n                        current_params[param] = np.random.uniform(min_val, max_val)\n            else:\n                current_params = initial_params.copy()\n            \n            current_score = self.evaluate_params(current_params)\n            \n            # Hill climbing\n            for iteration in range(max_iterations):\n                improved = False\n                \n                # Try modifying each parameter\n                for param in current_params:\n                    if param not in param_space:\n                        continue\n                    \n                    min_val, max_val = param_space[param]\n                    \n                    # Try multiple step sizes\n                    step_sizes = [0.01, 0.05, 0.1] if isinstance(min_val, float) else [1, 2, 5]\n                    \n                    for step_size in step_sizes:\n                        for direction in [-1, 1]:\n                            new_params = current_params.copy()\n                            \n                            if isinstance(current_params[param], int):\n                                new_params[param] = current_params[param] + direction * step_size\n                                new_params[param] = int(np.clip(new_params[param], min_val, max_val))\n                            else:\n                                delta = step_size * (max_val - min_val)\n                                new_params[param] = current_params[param] + direction * delta\n                                new_params[param] = np.clip(new_params[param], min_val, max_val)\n                            \n                            new_score = self.evaluate_params(new_params)\n                            \n                            if new_score > current_score:\n                                current_score = new_score\n                                current_params = new_params.copy()\n                                improved = True\n                                break\n                        \n                        if improved:\n                            break\n                    \n                    if improved:\n                        break\n                \n                if not improved:\n                    print(f\"  Converged at iteration {iteration}\")\n                    break\n            \n            # Update global best\n            if current_score > global_best_score:\n                global_best_score = current_score\n                global_best_params = current_params.copy()\n                print(f\"  New global best! Score: {global_best_score:.4f}\")\n        \n        return global_best_params\n    \n    def bayesian_optimization(self, param_space: Dict, n_trials: int = 50):\n        \"\"\"\n        Use Optuna for Bayesian optimization\n        \"\"\"\n        print(\"\\nStarting Bayesian optimization...\")\n        \n        def objective(trial):\n            params = {\n                'n_estimators': 500,  # Fixed\n                'tree_method': 'hist',\n                'random_state': 42,\n                'n_jobs': -1,\n                'verbosity': 0\n            }\n            \n            # Sample parameters\n            for param, (min_val, max_val) in param_space.items():\n                if param == 'n_estimators':\n                    continue\n                    \n                if isinstance(min_val, int):\n                    params[param] = trial.suggest_int(param, min_val, max_val)\n                else:\n                    params[param] = trial.suggest_float(param, min_val, max_val)\n            \n            return self.evaluate_params(params)\n        \n        # Create study\n        study = optuna.create_study(direction='maximize', seed=42)\n        study.optimize(objective, n_trials=n_trials, show_progress_bar=True)\n        \n        # Get best parameters\n        best_params = study.best_params\n        best_params.update({\n            'n_estimators': 500,\n            'tree_method': 'hist',\n            'random_state': 42,\n            'n_jobs': -1,\n            'verbosity': 0\n        })\n        \n        return best_params\n    \n    def analyze_overfitting_curve(self):\n        \"\"\"\n        Analyze the relationship between parameters and overfitting\n        \"\"\"\n        if not self.performance_history:\n            return None\n        \n        df = pd.DataFrame(self.performance_history)\n        \n        # Find parameters that correlate with overfitting\n        overfitting_correlations = {}\n        for param in df['params'].iloc[0].keys():\n            if param in ['tree_method', 'random_state', 'n_jobs', 'verbosity']:\n                continue\n            \n            param_values = [p[param] for p in df['params']]\n            gaps = df['gap'].values\n            \n            if len(set(param_values)) > 1:  # Only if parameter varied\n                correlation = pearsonr(param_values, gaps)[0]\n                overfitting_correlations[param] = correlation\n        \n        return overfitting_correlations\n    \n    def create_anti_overfit_ensemble(self, param_space: Dict):\n        \"\"\"\n        Create ensemble with models that overfit in opposite directions\n        based on learned parameter-overfitting relationships\n        \"\"\"\n        print(\"\\nCreating anti-overfitting ensemble...\")\n        \n        # Analyze overfitting patterns\n        overfit_correlations = self.analyze_overfitting_curve()\n        \n        if not overfit_correlations:\n            print(\"No overfitting analysis available\")\n            return None\n        \n        # Create models that overfit in opposite directions\n        models = []\n        \n        # Model 1: Parameters that increase overfitting\n        overfit_params = self.best_params.copy()\n        for param, correlation in overfit_correlations.items():\n            if correlation > 0.3 and param in param_space:  # Increases overfitting\n                min_val, max_val = param_space[param]\n                if isinstance(overfit_params[param], int):\n                    overfit_params[param] = int(min_val + 0.8 * (max_val - min_val))\n                else:\n                    overfit_params[param] = min_val + 0.8 * (max_val - min_val)\n        \n        # Model 2: Parameters that decrease overfitting\n        underfit_params = self.best_params.copy()\n        for param, correlation in overfit_correlations.items():\n            if correlation > 0.3 and param in param_space:  # Increases overfitting\n                min_val, max_val = param_space[param]\n                if isinstance(underfit_params[param], int):\n                    underfit_params[param] = int(min_val + 0.2 * (max_val - min_val))\n                else:\n                    underfit_params[param] = min_val + 0.2 * (max_val - min_val)\n        \n        # Train models\n        print(\"Training overfit model...\")\n        model_overfit = XGBRegressor(**overfit_params)\n        model_overfit.fit(self.X_train, self.y_train)\n        \n        print(\"Training underfit model...\")\n        model_underfit = XGBRegressor(**underfit_params)\n        model_underfit.fit(self.X_train, self.y_train)\n        \n        print(\"Training balanced model...\")\n        model_balanced = XGBRegressor(**self.best_params)\n        model_balanced.fit(self.X_train, self.y_train)\n        \n        # Return ensemble predictions\n        pred_overfit = model_overfit.predict(self.X_test)\n        pred_underfit = model_underfit.predict(self.X_test)\n        pred_balanced = model_balanced.predict(self.X_test)\n        \n        # Weighted average based on CV performance\n        weights = [0.25, 0.25, 0.5]  # Can be optimized\n        ensemble_pred = (\n            weights[0] * pred_overfit + \n            weights[1] * pred_underfit + \n            weights[2] * pred_balanced\n        )\n        \n        return ensemble_pred, {\n            'overfit': model_overfit,\n            'underfit': model_underfit,\n            'balanced': model_balanced\n        }\n\n# Main pipeline\ndef reduce_mem_usage(df, name=\"\"):\n    print(f\"Optimizing memory for {name}...\")\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage: {start_mem:.2f} MB -> {end_mem:.2f} MB ({100*(start_mem-end_mem)/start_mem:.1f}% reduction)')\n    return df\n\ndef add_features(df):\n    \"\"\"Create features for market microstructure\"\"\"\n    print(\"Engineering features...\")\n    \n    # Basic features\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    df['log_volume'] = np.log1p(df['volume'])\n    \n    # Handle infinities\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    return df\n\n# Load data\nprint(\"\\nLoading data...\")\ntrain = pd.read_parquet(CFG.train_path)\ntest = pd.read_parquet(CFG.test_path)\nsubmission = pd.read_csv(CFG.sample_sub_path)\n\n# Feature engineering\ntrain = add_features(train)\ntest = add_features(test)\n\n# Memory optimization\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\n# Select features\nselected_x_features = [\"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\"]\navailable_x_features = [f for f in selected_x_features if f in train.columns]\n\n# Add more X features\nif len(available_x_features) < 30:\n    all_x_features = [col for col in train.columns if col.startswith('X') and col[1:].isdigit()]\n    additional_x = [f for f in all_x_features if f not in available_x_features][:30]\n    available_x_features.extend(additional_x)\n\nmarket_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\nengineered_features = ['bid_ask_spread', 'bid_ask_ratio', 'buy_sell_ratio', \n                      'order_flow_imbalance', 'total_liquidity', 'log_volume']\n\nselected_features = market_features + available_x_features + engineered_features\nselected_features = [f for f in selected_features if f in train.columns]\n\n# Prepare data\nX_train = train[selected_features]\ny_train = train['label']\nX_test = test[selected_features]\n\n# Scaling\nscaler = StandardScaler()\nX_train_scaled = pd.DataFrame(\n    scaler.fit_transform(X_train),\n    columns=selected_features,\n    index=X_train.index\n)\nX_test_scaled = pd.DataFrame(\n    scaler.transform(X_test),\n    columns=selected_features,\n    index=X_test.index\n)\n\n# Define parameter space\nparam_space = {\n    'max_depth': (3, 12),\n    'learning_rate': (0.001, 0.1),\n    'subsample': (0.5, 1.0),\n    'colsample_bytree': (0.3, 1.0),\n    'min_child_weight': (1, 50),\n    'reg_alpha': (0.0, 10.0),\n    'reg_lambda': (0.0, 10.0),\n    'gamma': (0.0, 5.0)\n}\n\n# Initialize optimizer\noptimizer = IntelligentXGBOptimizer(X_train_scaled, y_train, X_test_scaled)\n\n# Try different optimization strategies\nprint(\"\\n\" + \"=\"*80)\nprint(\"OPTIMIZATION PHASE\")\nprint(\"=\"*80)\n\n# 1. Adaptive Grid Search\nbest_params_adaptive = optimizer.adaptive_grid_search(param_space, n_iterations=15)\nprint(f\"\\nAdaptive search best score: {optimizer.best_cv_score:.4f}\")\n\n# 2. Hill Climbing\ninitial_params = {\n    'max_depth': 6,\n    'learning_rate': 0.02,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'min_child_weight': 10,\n    'reg_alpha': 1.0,\n    'reg_lambda': 1.0,\n    'gamma': 0.1,\n    'n_estimators': 500,\n    'tree_method': 'hist',\n    'random_state': 42,\n    'n_jobs': -1,\n    'verbosity': 0\n}\n\nbest_params_hill = optimizer.hill_climbing_optimizer(\n    initial_params, param_space, n_restarts=2, max_iterations=20\n)\n\n# 3. Bayesian Optimization (if time permits)\n# best_params_bayes = optimizer.bayesian_optimization(param_space, n_trials=30)\n\n# Create anti-overfitting ensemble\nensemble_pred, models = optimizer.create_anti_overfit_ensemble(param_space)\n\n# Analyze results\nprint(\"\\n\" + \"=\"*80)\nprint(\"ANALYSIS PHASE\")\nprint(\"=\"*80)\n\n# Show overfitting analysis\noverfit_correlations = optimizer.analyze_overfitting_curve()\nif overfit_correlations:\n    print(\"\\nParameter-Overfitting Correlations:\")\n    for param, corr in sorted(overfit_correlations.items(), key=lambda x: abs(x[1]), reverse=True):\n        print(f\"  {param}: {corr:.3f}\")\n\n# Show performance history\nhistory_df = pd.DataFrame(optimizer.performance_history)\nprint(\"\\nPerformance Summary:\")\nprint(f\"  Best CV Score: {history_df['cv_score'].max():.4f}\")\nprint(f\"  Best Train Score: {history_df['train_score'].max():.4f}\")\nprint(f\"  Minimum Gap: {history_df['gap'].min():.4f}\")\nprint(f\"  Average Gap: {history_df['gap'].mean():.4f}\")\n\n# Create submission\nsubmission['prediction'] = ensemble_pred\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"Submission saved to submission.csv\")\nprint(submission.head())\n\n# Save detailed results\nhistory_df.to_csv('optimization_history.csv', index=False)\nprint(\"\\nOptimization history saved to optimization_history.csv\")\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"Pipeline completed successfully!\")\nprint(\"=\"*80)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}