{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nDRW Crypto Market Prediction - Complete Solution with AutoML Options\n===================================================================\nIncludes Optuna, LightAutoML, and FLAML for comprehensive optimization\n\"\"\"\n\nimport gc\nimport os\nimport sys\nimport numpy as np\nimport pandas as pd\nimport xgboost as xgb\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.stats import pearsonr\nimport warnings\nimport time\nfrom datetime import datetime\n\nwarnings.filterwarnings('ignore')\n\n# Try to import optimization libraries\ntry:\n    import optuna\n    from optuna.samplers import TPESampler\n    optuna.logging.set_verbosity(optuna.logging.WARNING)\n    OPTUNA_AVAILABLE = True\nexcept ImportError:\n    OPTUNA_AVAILABLE = False\n    print(\"Optuna not available - will use default parameters\")\n\ntry:\n    from lightautoml.automl.presets.tabular_presets import TabularAutoML\n    from lightautoml.tasks import Task\n    LIGHTAUTOML_AVAILABLE = True\nexcept ImportError:\n    LIGHTAUTOML_AVAILABLE = False\n    print(\"LightAutoML not available\")\n\ntry:\n    from flaml import AutoML as FLAMLAutoML\n    FLAML_AVAILABLE = True\nexcept ImportError:\n    FLAML_AVAILABLE = False\n    print(\"FLAML not available\")\n\n# Set random seed\nnp.random.seed(42)\n\n# Check environment\nKAGGLE_INPUT_PATH = '/kaggle/input/drw-crypto-market-prediction'\nDATA_PATH = KAGGLE_INPUT_PATH if os.path.exists(KAGGLE_INPUT_PATH) else '.'\n\n# =========================\n# Configuration\n# =========================\nclass Config:\n    TRAIN_PATH = os.path.join(DATA_PATH, \"train.parquet\")\n    TEST_PATH = os.path.join(DATA_PATH, \"test.parquet\")\n    SUBMISSION_PATH = os.path.join(DATA_PATH, \"sample_submission.csv\")\n    \n    # Core features - proven to work well\n    CORE_FEATURES = [\n        \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \n        \"X860\", \"X674\", \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \n        \"X178\", \"X532\", \"X168\", \"X612\", \"X888\", \"X421\", \"X333\"\n    ]\n    \n    MARKET_FEATURES = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    \n    # Features for dual representation\n    DUAL_FEATURES = [\"X612\", \"X860\", \"X168\", \"X174\", \"X333\", \"X345\"]\n    \n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    RANDOM_STATE = 42\n    \n    # Memory optimization\n    USE_FLOAT32 = True\n    \n    # Sampling strategy\n    SELECTION_SAMPLES = [100000, 200000, 300000]\n    SELECTION_RATIOS = [0.8, 0.7, 0.6]\n    MAX_ADDITIONAL_FEATURES = 50\n    \n    # Tuning parameters\n    USE_AUTOML = True  # Use AutoML if available\n    N_TUNING_TRIALS = 20 if OPTUNA_AVAILABLE else 0\n    TUNING_SAMPLE_SIZE = 100000\n    \n    # AutoML settings\n    LIGHTAUTOML_TIMEOUT = 300  # 5 minutes\n    FLAML_TIME_BUDGET = 300    # 5 minutes\n\n# Default parameters if optimization is not available\nDEFAULT_XGB_PARAMS = {\n    'tree_method': 'hist',\n    'device': 'cpu',\n    'n_estimators': 500,\n    'max_depth': 10,\n    'learning_rate': 0.03,\n    'subsample': 0.7,\n    'colsample_bytree': 0.7,\n    'gamma': 1.5,\n    'reg_alpha': 10,\n    'reg_lambda': 10,\n    'random_state': Config.RANDOM_STATE,\n    'n_jobs': 2,\n    'verbosity': 0\n}\n\nDEFAULT_LGB_PARAMS = {\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse',\n    'n_estimators': 500,\n    'num_leaves': 31,\n    'learning_rate': 0.03,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 0.1,\n    'reg_lambda': 0.1,\n    'min_child_samples': 50,\n    'random_state': Config.RANDOM_STATE,\n    'n_jobs': 2,\n    'verbose': -1\n}\n\n# =========================\n# Memory Management\n# =========================\ndef reduce_mem_usage(dataframe, verbose=True):\n    \"\"\"Optimized memory reduction function\"\"\"\n    if verbose:\n        print('Reducing memory usage...')\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n        if col_type != 'object':\n            c_min = dataframe[col].min()\n            c_max = dataframe[col].max()\n            \n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    dataframe[col] = dataframe[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    dataframe[col] = dataframe[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    dataframe[col] = dataframe[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    dataframe[col] = dataframe[col].astype(np.int64)\n            else:\n                if Config.USE_FLOAT32 and c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    dataframe[col] = dataframe[col].astype(np.float32)\n                elif c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    dataframe[col] = dataframe[col].astype(np.float16)\n                else:\n                    dataframe[col] = dataframe[col].astype(np.float32)\n    \n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    if verbose:\n        print(f'--- Memory usage before: {initial_mem_usage:.2f} MB')\n        print(f'--- Memory usage after: {final_mem_usage:.2f} MB')\n        print(f'--- Decreased by {100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage:.1f}%\\n')\n    \n    return dataframe\n\ndef print_memory_usage():\n    \"\"\"Print current memory usage\"\"\"\n    try:\n        import psutil\n        process = psutil.Process(os.getpid())\n        mem_info = process.memory_info()\n        print(f\"Current memory usage: {mem_info.rss / 1024 / 1024:.2f} MB\")\n    except:\n        pass\n\ndef clean_memory():\n    \"\"\"Force garbage collection\"\"\"\n    gc.collect()\n\n# =========================\n# Feature Engineering\n# =========================\nclass MemoryEfficientFeatureEngineer:\n    \"\"\"Memory-efficient feature engineering\"\"\"\n    \n    def __init__(self):\n        self.selected_features = None\n        self.feature_stats = {}\n    \n    def create_rank_features_batch(self, df, features, batch_size=10):\n        \"\"\"Create rank features in batches to save memory\"\"\"\n        rank_features = []\n        \n        for i in range(0, len(features), batch_size):\n            batch_features = features[i:i+batch_size]\n            batch_df = pd.DataFrame(index=df.index)\n            \n            for feature in batch_features:\n                if feature in df.columns:\n                    batch_df[f\"{feature}_rank\"] = df[feature].rank(method='dense', pct=True).astype(np.float32)\n            \n            rank_features.append(batch_df)\n            clean_memory()\n        \n        if rank_features:\n            return pd.concat(rank_features, axis=1)\n        else:\n            return pd.DataFrame(index=df.index)\n    \n    def create_essential_features(self, df):\n        \"\"\"Create only the most essential engineered features\"\"\"\n        features = pd.DataFrame(index=df.index)\n        \n        # Core market microstructure features\n        if all(col in df.columns for col in ['bid_qty', 'ask_qty']):\n            total = df['bid_qty'] + df['ask_qty'] + 1e-8\n            features['bid_ask_imbalance'] = ((df['bid_qty'] - df['ask_qty']) / total).astype(np.float32)\n            features['bid_ratio'] = (df['bid_qty'] / total).astype(np.float32)\n        \n        if all(col in df.columns for col in ['buy_qty', 'sell_qty']):\n            total = df['buy_qty'] + df['sell_qty'] + 1e-8\n            features['buy_sell_pressure'] = ((df['buy_qty'] - df['sell_qty']) / total).astype(np.float32)\n        \n        if 'volume' in df.columns and all(col in df.columns for col in ['bid_qty', 'ask_qty']):\n            features['liquidity'] = ((df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-8)).astype(np.float32)\n        \n        return features\n    \n    def create_top_interactions(self, df, top_features, n_interactions=10):\n        \"\"\"Create interactions only between top features\"\"\"\n        interactions = pd.DataFrame(index=df.index)\n        \n        # Only use top 5 features for interactions\n        interaction_features = [f for f in top_features if f in df.columns][:5]\n        \n        count = 0\n        for i in range(len(interaction_features)):\n            for j in range(i+1, len(interaction_features)):\n                if count >= n_interactions:\n                    break\n                f1, f2 = interaction_features[i], interaction_features[j]\n                interactions[f'{f1}_x_{f2}'] = (df[f1] * df[f2]).astype(np.float32)\n                count += 1\n        \n        return interactions\n\n# =========================\n# Progressive Feature Selection\n# =========================\nclass OptimizedFeatureSelector:\n    \"\"\"Memory-efficient feature selection\"\"\"\n    \n    def __init__(self):\n        self.selected_features = None\n        self.feature_scores = None\n    \n    def evaluate_features_chunked(self, train_df, candidate_features, label_col, chunk_size=50):\n        \"\"\"Evaluate features in chunks to save memory\"\"\"\n        scores = {}\n        \n        for i in range(0, len(candidate_features), chunk_size):\n            chunk_features = candidate_features[i:i+chunk_size]\n            \n            # Calculate correlations\n            for feat in chunk_features:\n                if feat in train_df.columns:\n                    corr = abs(train_df[feat].corr(train_df[label_col]))\n                    if not np.isnan(corr):\n                        scores[feat] = corr\n            \n            clean_memory()\n        \n        return scores\n    \n    def progressive_selection(self, train_df, baseline_features, max_features=50):\n        \"\"\"Progressive feature selection with larger samples\"\"\"\n        print(\"\\n=== Progressive Feature Selection ===\")\n        \n        # Get candidate features\n        all_features = [col for col in train_df.columns if col.startswith('X')]\n        candidates = [f for f in all_features if f not in baseline_features]\n        \n        if not candidates:\n            return []\n        \n        print(f\"Evaluating {len(candidates)} candidate features...\")\n        current_features = candidates\n        \n        # Progressive selection with increasing samples\n        for stage, (sample_size, keep_ratio) in enumerate(zip(Config.SELECTION_SAMPLES, Config.SELECTION_RATIOS), 1):\n            if len(current_features) <= max_features:\n                break\n            \n            # Use minimum of sample size and available data\n            actual_sample_size = min(sample_size, len(train_df))\n            print(f\"\\nStage {stage}: Using {actual_sample_size:,} samples (keeping {keep_ratio:.0%})\")\n            \n            # Sample recent data\n            start_idx = max(0, len(train_df) - actual_sample_size)\n            sample_df = train_df.iloc[start_idx:start_idx + actual_sample_size]\n            \n            # Evaluate features\n            scores = self.evaluate_features_chunked(sample_df, current_features, Config.LABEL_COLUMN)\n            \n            # Sort by score\n            sorted_features = sorted(scores.items(), key=lambda x: x[1], reverse=True)\n            \n            # Keep top features\n            n_keep = int(len(sorted_features) * keep_ratio)\n            n_keep = min(n_keep, max_features)\n            current_features = [feat for feat, _ in sorted_features[:n_keep]]\n            \n            print(f\"  Kept {len(current_features)} features (top score: {sorted_features[0][1]:.4f})\")\n            \n            clean_memory()\n        \n        # Final selection\n        self.selected_features = current_features[:max_features]\n        self.feature_scores = {f: s for f, s in scores.items() if f in self.selected_features}\n        \n        print(f\"\\nFinal selection: {len(self.selected_features)} features\")\n        return self.selected_features\n\n# =========================\n# Hyperparameter Tuning with Multiple Options\n# =========================\nclass ModelTuner:\n    \"\"\"Hyperparameter tuning with Optuna\"\"\"\n    \n    def __init__(self, model_type='xgb'):\n        self.model_type = model_type\n        self.best_params = None\n    \n    def objective(self, trial, X_train, y_train, X_valid, y_valid):\n        \"\"\"Objective function for Optuna\"\"\"\n        \n        if self.model_type == 'xgb':\n            params = {\n                'tree_method': 'hist',\n                'device': 'cpu',\n                'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n                'max_depth': trial.suggest_int('max_depth', 3, 20),\n                'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n                'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n                'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n                'gamma': trial.suggest_float('gamma', 0, 5),\n                'reg_alpha': trial.suggest_float('reg_alpha', 0, 100),\n                'reg_lambda': trial.suggest_float('reg_lambda', 0, 100),\n                'random_state': Config.RANDOM_STATE,\n                'n_jobs': 2,\n                'verbosity': 0\n            }\n            \n            model = XGBRegressor(**params)\n            model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], \n                     verbose=False, early_stopping_rounds=50)\n            \n        else:  # lgb\n            params = {\n                'boosting_type': 'gbdt',\n                'objective': 'regression',\n                'metric': 'rmse',\n                'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n                'num_leaves': trial.suggest_int('num_leaves', 10, 100),\n                'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n                'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n                'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n                'reg_alpha': trial.suggest_float('reg_alpha', 0, 10),\n                'reg_lambda': trial.suggest_float('reg_lambda', 0, 10),\n                'min_child_samples': trial.suggest_int('min_child_samples', 10, 100),\n                'random_state': Config.RANDOM_STATE,\n                'n_jobs': 2,\n                'verbose': -1\n            }\n            \n            model = LGBMRegressor(**params)\n            model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)])\n        \n        # Predict and calculate score\n        y_pred = model.predict(X_valid)\n        score = pearsonr(y_valid, y_pred)[0]\n        \n        return score\n    \n    def tune(self, train_df, features, n_trials=20):\n        \"\"\"Run hyperparameter tuning\"\"\"\n        if not OPTUNA_AVAILABLE:\n            print(f\"  Optuna not available - using default {self.model_type.upper()} parameters\")\n            if self.model_type == 'xgb':\n                return DEFAULT_XGB_PARAMS\n            else:\n                return DEFAULT_LGB_PARAMS\n        \n        print(f\"\\nTuning {self.model_type.upper()} hyperparameters with Optuna...\")\n        \n        # Use a sample for tuning\n        sample_size = min(Config.TUNING_SAMPLE_SIZE, len(train_df))\n        sample_df = train_df.tail(sample_size)\n        \n        # Split data\n        X = sample_df[features].values\n        y = sample_df[Config.LABEL_COLUMN].values\n        \n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X, y, test_size=0.2, random_state=Config.RANDOM_STATE\n        )\n        \n        # Create study\n        study = optuna.create_study(\n            direction='maximize',\n            sampler=TPESampler(seed=Config.RANDOM_STATE)\n        )\n        \n        # Optimize\n        study.optimize(\n            lambda trial: self.objective(trial, X_train, y_train, X_valid, y_valid),\n            n_trials=n_trials\n        )\n        \n        self.best_params = study.best_params\n        print(f\"Best score: {study.best_value:.4f}\")\n        \n        # Add fixed parameters\n        if self.model_type == 'xgb':\n            self.best_params.update({\n                'tree_method': 'hist',\n                'device': 'cpu',\n                'random_state': Config.RANDOM_STATE,\n                'n_jobs': 2,\n                'verbosity': 0\n            })\n        else:\n            self.best_params.update({\n                'boosting_type': 'gbdt',\n                'objective': 'regression',\n                'metric': 'rmse',\n                'random_state': Config.RANDOM_STATE,\n                'n_jobs': 2,\n                'verbose': -1\n            })\n        \n        return self.best_params\n\n# =========================\n# AutoML Training Functions\n# =========================\ndef train_with_lightautoml(train_df, test_df, features):\n    \"\"\"Train using LightAutoML - Fixed version\"\"\"\n    if not LIGHTAUTOML_AVAILABLE:\n        return None, 0\n    \n    print(\"\\nTraining with LightAutoML...\")\n    \n    # Prepare data - LightAutoML expects a single DataFrame with target column\n    train_data = train_df[features + [Config.LABEL_COLUMN]].copy()\n    test_data = test_df[features].copy()\n    \n    # Create AutoML task\n    task = Task('reg', metric='r2')\n    \n    # Initialize AutoML\n    automl = TabularAutoML(\n        task=task,\n        timeout=Config.LIGHTAUTOML_TIMEOUT,\n        cpu_limit=2,\n        reader_params={'n_jobs': 2}\n    )\n    \n    # Train - pass the full DataFrame and specify roles\n    oof_pred = automl.fit_predict(\n        train_data,\n        roles={'target': Config.LABEL_COLUMN}\n    )\n    \n    # Score\n    score = pearsonr(train_df[Config.LABEL_COLUMN], oof_pred.data[:, 0])[0]\n    print(f\"LightAutoML CV Score: {score:.4f}\")\n    \n    # Test predictions\n    test_pred = automl.predict(test_data).data[:, 0]\n    \n    return test_pred, score\n\ndef train_with_flaml(train_df, test_df, features):\n    \"\"\"Train using FLAML AutoML\"\"\"\n    if not FLAML_AVAILABLE:\n        return None, 0\n    \n    print(\"\\nTraining with FLAML AutoML...\")\n    \n    # Prepare data\n    X_train = train_df[features].values\n    y_train = train_df[Config.LABEL_COLUMN].values\n    X_test = test_df[features].values\n    \n    # Initialize FLAML\n    automl = FLAMLAutoML()\n    \n    # Configure settings\n    settings = {\n        \"time_budget\": Config.FLAML_TIME_BUDGET,\n        \"metric\": 'r2',\n        \"task\": 'regression',\n        \"n_jobs\": 2,\n        \"estimator_list\": ['xgboost', 'lgbm', 'rf', 'extra_tree'],\n        \"seed\": Config.RANDOM_STATE\n    }\n    \n    # Train\n    automl.fit(X_train, y_train, **settings)\n    \n    # Get CV score - FLAML uses negative R2, so we need to negate it\n    score = -automl.best_score if automl.best_score < 0 else automl.best_score\n    print(f\"FLAML Best CV Score: {score:.4f}\")\n    print(f\"FLAML Best Estimator: {automl.best_estimator}\")\n    \n    # Test predictions\n    test_pred = automl.predict(X_test)\n    \n    return test_pred, score\n\n# =========================\n# Model Training with Time Decay\n# =========================\ndef train_optimized_model(train_df, test_df, features, params, model_type='xgb', use_time_decay=True):\n    \"\"\"Train model with optimized parameters\"\"\"\n    print(f\"\\nTraining {model_type.upper()} with optimized parameters...\")\n    \n    n_samples = len(train_df)\n    oof_predictions = np.zeros(n_samples)\n    test_predictions = np.zeros(len(test_df))\n    \n    # Time decay weights\n    if use_time_decay:\n        positions = np.arange(n_samples)\n        weights = 0.95 ** (1.0 - positions / (n_samples - 1))\n        weights = weights * n_samples / weights.sum()\n    else:\n        weights = np.ones(n_samples)\n    \n    kf = KFold(n_splits=Config.N_FOLDS, shuffle=False)\n    fold_scores = []\n    \n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), 1):\n        print(f\"  Fold {fold}/{Config.N_FOLDS}\", end=' ')\n        \n        X_train = train_df.iloc[train_idx][features].values\n        y_train = train_df.iloc[train_idx][Config.LABEL_COLUMN].values\n        X_valid = train_df.iloc[valid_idx][features].values\n        y_valid = train_df.iloc[valid_idx][Config.LABEL_COLUMN].values\n        w_train = weights[train_idx]\n        \n        # Train model\n        if model_type == 'xgb':\n            model = XGBRegressor(**params)\n            model.fit(X_train, y_train, sample_weight=w_train,\n                     eval_set=[(X_valid, y_valid)], verbose=False)\n        else:  # lgb\n            model = LGBMRegressor(**params)\n            model.fit(X_train, y_train, sample_weight=w_train,\n                     eval_set=[(X_valid, y_valid)])\n        \n        # Predictions\n        oof_predictions[valid_idx] = model.predict(X_valid)\n        test_predictions += model.predict(test_df[features].values) / Config.N_FOLDS\n        \n        # Score\n        fold_score = pearsonr(y_valid, oof_predictions[valid_idx])[0]\n        fold_scores.append(fold_score)\n        print(f\"Score: {fold_score:.4f}\")\n        \n        # Clean up\n        del model, X_train, y_train, X_valid, y_valid\n        clean_memory()\n    \n    # Overall score\n    cv_score = pearsonr(train_df[Config.LABEL_COLUMN].values, oof_predictions)[0]\n    print(f\"  CV Score: {cv_score:.4f} (std: {np.std(fold_scores):.4f})\")\n    \n    return test_predictions, cv_score\n\n# =========================\n# Main Pipeline\n# =========================\ndef main():\n    \"\"\"Main execution pipeline\"\"\"\n    print(\"=\" * 80)\n    print(\"DRW CRYPTO - OPTIMIZED SOLUTION WITH AUTOML\")\n    print(f\"Started at: {datetime.now()}\")\n    print(\"=\" * 80)\n    \n    # Show available libraries\n    print(\"\\nAvailable optimization libraries:\")\n    print(f\"  - Optuna: {'Yes' if OPTUNA_AVAILABLE else 'No'}\")\n    print(f\"  - LightAutoML: {'Yes' if LIGHTAUTOML_AVAILABLE else 'No'}\")\n    print(f\"  - FLAML: {'Yes' if FLAML_AVAILABLE else 'No'}\")\n    \n    # Initial memory status\n    print_memory_usage()\n    \n    # Load data\n    print(\"\\n1. Loading data...\")\n    train_df = pd.read_parquet(Config.TRAIN_PATH)\n    test_df = pd.read_parquet(Config.TEST_PATH)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n    \n    print(f\"Train shape: {train_df.shape}, Test shape: {test_df.shape}\")\n    \n    # Optimize memory immediately\n    print(\"\\n2. Optimizing memory...\")\n    train_df = reduce_mem_usage(train_df)\n    test_df = reduce_mem_usage(test_df)\n    print_memory_usage()\n    \n    # Initialize components\n    feature_engineer = MemoryEfficientFeatureEngineer()\n    feature_selector = OptimizedFeatureSelector()\n    \n    # Feature selection\n    print(\"\\n3. Feature engineering...\")\n    \n    # Select additional features using progressive selection\n    additional_features = feature_selector.progressive_selection(\n        train_df, Config.CORE_FEATURES, Config.MAX_ADDITIONAL_FEATURES\n    )\n    \n    # Prepare features efficiently\n    print(\"\\nPreparing final feature set...\")\n    \n    # Base features\n    base_features = [f for f in Config.CORE_FEATURES + Config.MARKET_FEATURES if f in train_df.columns]\n    \n    # Create engineered features\n    train_eng = feature_engineer.create_essential_features(train_df)\n    test_eng = feature_engineer.create_essential_features(test_df)\n    eng_feature_names = list(train_eng.columns)\n    \n    # Create rank features for dual representation (in batches)\n    dual_features = [f for f in Config.DUAL_FEATURES if f in train_df.columns]\n    train_dual_rank = feature_engineer.create_rank_features_batch(train_df, dual_features)\n    test_dual_rank = feature_engineer.create_rank_features_batch(test_df, dual_features)\n    \n    # Add selected additional features as ranks\n    if additional_features:\n        train_add_rank = feature_engineer.create_rank_features_batch(train_df, additional_features[:20])\n        test_add_rank = feature_engineer.create_rank_features_batch(test_df, additional_features[:20])\n    else:\n        train_add_rank = pd.DataFrame(index=train_df.index)\n        test_add_rank = pd.DataFrame(index=test_df.index)\n    \n    # Create minimal interactions\n    train_interactions = feature_engineer.create_top_interactions(train_df, Config.CORE_FEATURES, n_interactions=5)\n    test_interactions = feature_engineer.create_top_interactions(test_df, Config.CORE_FEATURES, n_interactions=5)\n    \n    # Combine all features\n    train_final = pd.concat([\n        train_df[base_features],\n        train_eng,\n        train_dual_rank,\n        train_add_rank,\n        train_interactions\n    ], axis=1)\n    train_final[Config.LABEL_COLUMN] = train_df[Config.LABEL_COLUMN]\n    \n    test_final = pd.concat([\n        test_df[base_features],\n        test_eng,\n        test_dual_rank,\n        test_add_rank,\n        test_interactions\n    ], axis=1)\n    \n    # Clean up intermediate dataframes\n    del train_eng, test_eng, train_dual_rank, test_dual_rank, train_add_rank, test_add_rank, train_interactions, test_interactions\n    clean_memory()\n    \n    # Get feature columns\n    feature_cols = [col for col in train_final.columns if col != Config.LABEL_COLUMN]\n    print(f\"\\nTotal features: {len(feature_cols)}\")\n    print(f\"  - Base features: {len(base_features)}\")\n    print(f\"  - Engineered features: {len(eng_feature_names)}\")\n    print(f\"  - Dual representation: {len(dual_features)}\")\n    print(f\"  - Additional features: {min(20, len(additional_features))}\")\n    print_memory_usage()\n    \n    # Store all predictions\n    all_predictions = {}\n    all_scores = {}\n    \n    # 4. Traditional models with hyperparameter tuning\n    print(\"\\n4. Traditional model training...\")\n    \n    # Tune XGBoost\n    xgb_tuner = ModelTuner(model_type='xgb')\n    xgb_params = xgb_tuner.tune(train_final, feature_cols, n_trials=Config.N_TUNING_TRIALS)\n    clean_memory()\n    \n    # Train XGBoost\n    xgb_pred, xgb_score = train_optimized_model(\n        train_final, test_final, feature_cols, xgb_params, \n        model_type='xgb', use_time_decay=True\n    )\n    all_predictions['xgb'] = xgb_pred\n    all_scores['xgb'] = xgb_score\n    clean_memory()\n    \n    # Tune LightGBM\n    lgb_tuner = ModelTuner(model_type='lgb')\n    lgb_params = lgb_tuner.tune(train_final, feature_cols, n_trials=Config.N_TUNING_TRIALS)\n    clean_memory()\n    \n    # Train LightGBM\n    lgb_pred, lgb_score = train_optimized_model(\n        train_final, test_final, feature_cols, lgb_params,\n        model_type='lgb', use_time_decay=True\n    )\n    all_predictions['lgb'] = lgb_pred\n    all_scores['lgb'] = lgb_score\n    clean_memory()\n    \n    # 5. AutoML models (if available and enabled)\n    if Config.USE_AUTOML:\n        print(\"\\n5. AutoML model training...\")\n        \n        # LightAutoML\n        try:\n            lama_pred, lama_score = train_with_lightautoml(train_final, test_final, feature_cols)\n            if lama_pred is not None:\n                all_predictions['lightautoml'] = lama_pred\n                all_scores['lightautoml'] = lama_score\n                clean_memory()\n        except Exception as e:\n            print(f\"LightAutoML failed: {e}\")\n        \n        # FLAML\n        try:\n            flaml_pred, flaml_score = train_with_flaml(train_final, test_final, feature_cols)\n            if flaml_pred is not None:\n                all_predictions['flaml'] = flaml_pred\n                all_scores['flaml'] = flaml_score\n                clean_memory()\n        except Exception as e:\n            print(f\"FLAML failed: {e}\")\n    \n    # 6. Create ensemble\n    print(\"\\n6. Creating ensemble...\")\n    \n    # Calculate weights based on scores\n    total_score = sum(all_scores.values())\n    weights = {name: score/total_score for name, score in all_scores.items()}\n    \n    # Create weighted ensemble\n    ensemble_pred = np.zeros_like(list(all_predictions.values())[0])\n    for name, pred in all_predictions.items():\n        ensemble_pred += weights[name] * pred\n    \n    print(\"\\nEnsemble weights:\")\n    for name, weight in weights.items():\n        print(f\"  {name}: {weight:.3f} (score: {all_scores[name]:.4f})\")\n    \n    # 7. Save predictions\n    print(\"\\n7. Saving predictions...\")\n    \n    # Save individual predictions\n    for name, pred in all_predictions.items():\n        submission = submission_df.copy()\n        submission['prediction'] = pred\n        submission.to_csv(f'submission_{name}.csv', index=False)\n        print(f\"Saved: submission_{name}.csv\")\n    \n    # Save ensemble\n    submission_ensemble = submission_df.copy()\n    submission_ensemble['prediction'] = ensemble_pred\n    submission_ensemble.to_csv('submission_ensemble.csv', index=False)\n    print(\"Saved: submission_ensemble.csv\")\n    \n    # Final summary\n    print(\"\\n\" + \"=\" * 80)\n    print(\"SUMMARY\")\n    print(\"=\" * 80)\n    print(f\"Features used: {len(feature_cols)}\")\n    print(f\"\\nModel scores:\")\n    for name, score in all_scores.items():\n        print(f\"  {name}: {score:.4f}\")\n    print(f\"\\nExpected ensemble performance: >{max(all_scores.values()):.4f}\")\n    \n    if OPTUNA_AVAILABLE and Config.N_TUNING_TRIALS > 0:\n        print(f\"\\nBest XGBoost params: {xgb_params}\")\n        print(f\"\\nBest LightGBM params: {lgb_params}\")\n    \n    print(f\"\\nCompleted at: {datetime.now()}\")\n    print_memory_usage()\n    \n    return ensemble_pred, feature_cols\n\n# =========================\n# Entry Point\n# =========================\nif __name__ == \"__main__\":\n    predictions, features = main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}