{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:07:42.039419Z","iopub.execute_input":"2025-05-28T20:07:42.039676Z","iopub.status.idle":"2025-05-28T20:07:42.403946Z","shell.execute_reply.started":"2025-05-28T20:07:42.03963Z","shell.execute_reply":"2025-05-28T20:07:42.403334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required packages\n!pip install torch==2.0.1 -q\n!pip install einops==0.7.0 -q\n!pip install scikit-learn pandas numpy xgboost psutil -q\n\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.feature_selection import VarianceThreshold, mutual_info_regression\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport time\nimport warnings\nimport json\nimport gc\nimport psutil\nimport os\nfrom einops import rearrange\nimport matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\n# Check if CUDA is available and set device accordingly\nif torch.cuda.is_available():\n    device = torch.device('cuda')\n    torch.cuda.empty_cache()  # Clear cache at start\n    # Set memory fraction to avoid OOM\n    torch.cuda.set_per_process_memory_fraction(0.8)\nelse:\n    device = torch.device('cpu')\n    \nprint(f\"Using device: {device}\")\n\nclass LightweightAttention(nn.Module):\n    \"\"\"Lightweight attention module optimized for memory efficiency.\"\"\"\n    \n    def __init__(self, d_model, n_heads=4, dropout=0.1):\n        super().__init__()\n        self.d_model = d_model\n        self.n_heads = n_heads\n        self.d_k = d_model // n_heads\n        \n        # Single projection for Q, K, V to save memory\n        self.qkv_proj = nn.Linear(d_model, 3 * d_model)\n        self.out_proj = nn.Linear(d_model, d_model)\n        \n        self.dropout = nn.Dropout(dropout)\n        self.layer_norm = nn.LayerNorm(d_model)\n        \n    def forward(self, x, return_attention=False):\n        batch_size, seq_len, _ = x.shape\n        \n        # Single projection and split\n        qkv = self.qkv_proj(x)\n        qkv = rearrange(qkv, 'b l (three h d) -> three b h l d', three=3, h=self.n_heads)\n        Q, K, V = qkv[0], qkv[1], qkv[2]\n        \n        # Scaled dot-product attention\n        scale = 1.0 / np.sqrt(self.d_k)\n        attention_scores = torch.matmul(Q, K.transpose(-2, -1)) * scale\n        attention_weights = torch.softmax(attention_scores, dim=-1)\n        attention_weights = self.dropout(attention_weights)\n        \n        # Apply attention\n        attention_output = torch.matmul(attention_weights, V)\n        attention_output = rearrange(attention_output, 'b h l d -> b l (h d)')\n        \n        # Output projection\n        output = self.out_proj(attention_output)\n        output = self.dropout(output)\n        \n        # Residual connection and layer norm\n        output = self.layer_norm(x + output)\n        \n        if return_attention:\n            return output, attention_weights\n        return output\n\nclass MemoryEfficientFTTransformer(nn.Module):\n    \"\"\"Memory-efficient Feature Tokenizer Transformer.\"\"\"\n    \n    def __init__(self, n_features, d_model=16, n_heads=2, n_layers=1, dropout=0.1):\n        super().__init__()\n        self.n_features = n_features\n        self.d_model = d_model\n        \n        # Shared feature embedding to save memory\n        self.shared_embedding = nn.Linear(1, d_model)\n        \n        # Learnable feature-specific biases instead of separate embeddings\n        self.feature_biases = nn.Parameter(torch.randn(n_features, d_model) * 0.02)\n        \n        # Single transformer layer to minimize memory\n        self.transformer = LightweightAttention(d_model, n_heads, dropout)\n        \n        # Lightweight output head\n        self.output_head = nn.Sequential(\n            nn.Linear(d_model * n_features, 64),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(64, 1)\n        )\n        \n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, x, return_attention=False):\n        batch_size = x.shape[0]\n        \n        # Shared embedding with feature-specific bias\n        feature_tokens = []\n        for i in range(self.n_features):\n            token = self.shared_embedding(x[:, i:i+1]) + self.feature_biases[i]\n            feature_tokens.append(token.unsqueeze(1))\n        \n        feature_tokens = torch.cat(feature_tokens, dim=1)\n        feature_tokens = self.dropout(feature_tokens)\n        \n        # Single transformer layer\n        if return_attention:\n            feature_tokens, attention_weights = self.transformer(feature_tokens, return_attention=True)\n        else:\n            feature_tokens = self.transformer(feature_tokens)\n        \n        # Output\n        output = feature_tokens.reshape(batch_size, -1)\n        output = self.output_head(output)\n        \n        if return_attention:\n            return output.squeeze(-1), [attention_weights]\n        return output.squeeze(-1)\n    \n    def get_feature_importance(self, x):\n        \"\"\"Extract feature importance from attention patterns.\"\"\"\n        with torch.no_grad():\n            _, attention_weights = self.forward(x, return_attention=True)\n            \n            # Average attention received by each feature\n            feature_importance = attention_weights[0].mean(dim=(0, 1, 2))\n            \n        return feature_importance.cpu().numpy()\n\nclass MemoryOptimizedTransformerSelector:\n    \"\"\"Memory-optimized transformer-based feature selection.\"\"\"\n    \n    def __init__(self, config):\n        self.config = config\n        self.feature_importance_scores = {}\n        self.selected_features = []\n        self.selection_results = None\n        \n    def get_memory_usage(self):\n        \"\"\"Monitor memory usage.\"\"\"\n        process = psutil.Process(os.getpid())\n        return process.memory_info().rss / 1024 / 1024 / 1024\n    \n    def log_memory(self, message):\n        \"\"\"Log current memory usage.\"\"\"\n        memory_gb = self.get_memory_usage()\n        available_gb = psutil.virtual_memory().available / 1024 / 1024 / 1024\n        print(f\"{message}: Using {memory_gb:.2f} GB, Available: {available_gb:.2f} GB\")\n        \n        if torch.cuda.is_available():\n            allocated = torch.cuda.memory_allocated() / 1024**3\n            reserved = torch.cuda.memory_reserved() / 1024**3\n            print(f\"  GPU: Allocated {allocated:.2f} GB, Reserved {reserved:.2f} GB\")\n    \n    def reduce_data_precision(self, df):\n        \"\"\"Optimize dataframe memory usage.\"\"\"\n        initial_mem = df.memory_usage().sum() / 1024**2\n        \n        for col in df.columns:\n            col_type = df[col].dtype\n            \n            if col_type != object:\n                c_min = df[col].min()\n                c_max = df[col].max()\n                \n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    else:\n                        df[col] = df[col].astype(np.int32)\n                else:\n                    df[col] = df[col].astype(np.float32)\n        \n        final_mem = df.memory_usage().sum() / 1024**2\n        reduction_pct = 100 * (initial_mem - final_mem) / initial_mem\n        print(f\"Memory reduced by {reduction_pct:.1f}% ({initial_mem:.1f} MB → {final_mem:.1f} MB)\")\n        \n        return df\n    \n    def validate_and_clean_data(self, df):\n        \"\"\"Validate and clean data.\"\"\"\n        print(\"Validating and cleaning data...\")\n        \n        numeric_cols = df.select_dtypes(include=[np.number]).columns\n        \n        # Handle infinities\n        inf_mask = np.isinf(df[numeric_cols].values)\n        if inf_mask.any():\n            print(f\"Found {inf_mask.sum()} infinite values, replacing with NaN\")\n            df[numeric_cols] = df[numeric_cols].replace([np.inf, -np.inf], np.nan)\n        \n        # Handle NaN values\n        nan_counts = df.isna().sum()\n        features_with_nan = nan_counts[nan_counts > 0]\n        if len(features_with_nan) > 0:\n            print(f\"Found NaN values in {len(features_with_nan)} features\")\n            for col in numeric_cols:\n                if df[col].isna().any():\n                    df[col] = df[col].fillna(df[col].median())\n        \n        # Clip extreme outliers\n        for col in numeric_cols:\n            col_data = df[col]\n            if col_data.std() > 0:\n                q1 = col_data.quantile(0.01)\n                q99 = col_data.quantile(0.99)\n                iqr = q99 - q1\n                \n                if iqr > 0:\n                    lower_bound = q1 - 3 * iqr\n                    upper_bound = q99 + 3 * iqr\n                    df[col] = col_data.clip(lower=lower_bound, upper=upper_bound)\n        \n        return df\n    \n    def aggressive_feature_prescreening(self, X, y, target_features=200):\n        \"\"\"Aggressively prescreen features using mutual information.\"\"\"\n        print(f\"\\nPrescreening features to top {target_features} using mutual information...\")\n        \n        # Calculate mutual information scores\n        mi_scores = mutual_info_regression(X, y, random_state=42, n_neighbors=5)\n        \n        # Get top features by mutual information\n        mi_df = pd.DataFrame({\n            'feature': X.columns,\n            'mi_score': mi_scores\n        }).sort_values('mi_score', ascending=False)\n        \n        top_features = mi_df.head(target_features)['feature'].tolist()\n        \n        print(f\"Selected top {len(top_features)} features by mutual information\")\n        print(f\"MI score range: {mi_df.iloc[0]['mi_score']:.4f} to {mi_df.iloc[target_features-1]['mi_score']:.4f}\")\n        \n        return X[top_features], top_features\n    \n    def remove_low_variance_features(self, X, threshold_percentile=20):\n        \"\"\"Remove low variance features more aggressively.\"\"\"\n        print(\"Removing low-variance features...\")\n        \n        variances = {}\n        zero_var_features = []\n        \n        for col in X.columns:\n            col_var = X[col].var()\n            if pd.isna(col_var) or col_var == 0:\n                zero_var_features.append(col)\n            else:\n                variances[col] = col_var\n        \n        if len(zero_var_features) > 0:\n            print(f\"Removing {len(zero_var_features)} zero-variance features\")\n            X = X.drop(columns=zero_var_features)\n        \n        if len(variances) == 0:\n            raise ValueError(\"All features have zero variance\")\n        \n        # More aggressive variance filtering\n        variance_values = list(variances.values())\n        threshold = np.percentile(variance_values, threshold_percentile)\n        \n        selected_features = [col for col, var in variances.items() if var > threshold]\n        removed_features = [col for col, var in variances.items() if var <= threshold] + zero_var_features\n        \n        print(f\"Removed {len(removed_features)} low-variance features\")\n        print(f\"Retained {len(selected_features)} features\")\n        \n        return X[selected_features], selected_features, removed_features\n    \n    def train_lightweight_transformer(self, X_train, y_train, feature_names, \n                                    epochs=20, batch_size=64, learning_rate=0.001):\n        \"\"\"Train lightweight transformer with minimal memory usage.\"\"\"\n        \n        # Clear GPU cache before training\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()\n        \n        # Scale data\n        scaler = RobustScaler()\n        X_scaled = scaler.fit_transform(X_train)\n        \n        # Convert to tensors with memory mapping for large datasets\n        X_tensor = torch.FloatTensor(X_scaled).to(device)\n        y_tensor = torch.FloatTensor(y_train.values).to(device)\n        \n        # Create data loader with small batch size\n        dataset = TensorDataset(X_tensor, y_tensor)\n        dataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)\n        \n        # Initialize lightweight model\n        n_features = X_scaled.shape[1]\n        model = MemoryEfficientFTTransformer(\n            n_features=n_features,\n            d_model=16,  # Very small embedding dimension\n            n_heads=2,   # Minimal attention heads\n            n_layers=1,  # Single layer\n            dropout=0.2\n        ).to(device)\n        \n        # Count parameters\n        total_params = sum(p.numel() for p in model.parameters())\n        print(f\"Model parameters: {total_params:,}\")\n        \n        # Loss and optimizer\n        criterion = nn.MSELoss()\n        optimizer = optim.AdamW(model.parameters(), lr=learning_rate, weight_decay=0.01)\n        scheduler = optim.lr_scheduler.OneCycleLR(\n            optimizer, \n            max_lr=learning_rate, \n            epochs=epochs,\n            steps_per_epoch=len(dataloader)\n        )\n        \n        # Training with gradient accumulation\n        model.train()\n        accumulation_steps = 4\n        \n        for epoch in range(epochs):\n            epoch_loss = 0\n            optimizer.zero_grad()\n            \n            for i, (batch_X, batch_y) in enumerate(dataloader):\n                predictions = model(batch_X)\n                loss = criterion(predictions, batch_y) / accumulation_steps\n                loss.backward()\n                \n                if (i + 1) % accumulation_steps == 0:\n                    torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n                    optimizer.step()\n                    optimizer.zero_grad()\n                \n                epoch_loss += loss.item() * accumulation_steps\n                scheduler.step()\n                \n                # Clear cache periodically\n                if (i + 1) % 50 == 0 and torch.cuda.is_available():\n                    torch.cuda.empty_cache()\n            \n            if (epoch + 1) % 5 == 0:\n                avg_loss = epoch_loss / len(dataloader)\n                print(f\"  Epoch {epoch+1}/{epochs}, Loss: {avg_loss:.6f}\")\n        \n        # Extract feature importance with smaller batch\n        model.eval()\n        importance_batch_size = min(500, len(X_tensor))\n        feature_importance = model.get_feature_importance(X_tensor[:importance_batch_size])\n        \n        # Create importance dictionary\n        importance_dict = {feature_names[i]: float(importance) \n                          for i, importance in enumerate(feature_importance)}\n        \n        # Clean up\n        del model, X_tensor, y_tensor, dataset, dataloader\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()\n        gc.collect()\n        \n        return importance_dict\n    \n    def ensemble_transformer_selection(self, X_train, y_train, feature_names, \n                                     n_models=2, sample_size=5000):\n        \"\"\"Train ensemble with very conservative parameters.\"\"\"\n        \n        all_importance_scores = []\n        \n        for i in range(n_models):\n            print(f\"\\nTraining lightweight transformer {i+1}/{n_models}...\")\n            \n            # Different random seed for diversity\n            np.random.seed(42 + i)\n            torch.manual_seed(42 + i)\n            \n            # Subsample\n            if len(X_train) > sample_size:\n                indices = np.random.choice(len(X_train), size=sample_size, replace=False)\n                X_sample = X_train.iloc[indices]\n                y_sample = y_train.iloc[indices]\n            else:\n                X_sample = X_train\n                y_sample = y_train\n            \n            try:\n                # Train model\n                importance_scores = self.train_lightweight_transformer(\n                    X_sample, y_sample, feature_names,\n                    epochs=20, batch_size=64\n                )\n                all_importance_scores.append(importance_scores)\n                \n            except RuntimeError as e:\n                if \"out of memory\" in str(e):\n                    print(\"GPU OOM detected, falling back to CPU\")\n                    # Force CPU usage\n                    global device\n                    device = torch.device('cpu')\n                    if torch.cuda.is_available():\n                        torch.cuda.empty_cache()\n                    \n                    # Retry with CPU\n                    importance_scores = self.train_lightweight_transformer(\n                        X_sample, y_sample, feature_names,\n                        epochs=15, batch_size=32\n                    )\n                    all_importance_scores.append(importance_scores)\n                else:\n                    raise e\n            \n            # Aggressive cleanup\n            gc.collect()\n            if torch.cuda.is_available():\n                torch.cuda.empty_cache()\n            self.log_memory(f\"After model {i+1}\")\n        \n        # Aggregate scores\n        aggregated_scores = {}\n        for feature in feature_names:\n            scores = [model_scores.get(feature, 0) for model_scores in all_importance_scores]\n            aggregated_scores[feature] = {\n                'mean': np.mean(scores),\n                'std': np.std(scores) if len(scores) > 1 else 0,\n                'min': np.min(scores),\n                'max': np.max(scores)\n            }\n        \n        return aggregated_scores\n    \n    def evaluate_feature_set(self, X_eval, y_eval, features):\n        \"\"\"Quick evaluation using XGBoost.\"\"\"\n        X_subset = X_eval[features]\n        \n        model = XGBRegressor(\n            n_estimators=30,\n            max_depth=3,\n            learning_rate=0.1,\n            random_state=42,\n            n_jobs=-1,\n            tree_method='hist'\n        )\n        \n        # 2-fold CV for speed\n        kf = KFold(n_splits=2, shuffle=True, random_state=42)\n        scores = []\n        \n        for train_idx, val_idx in kf.split(X_subset):\n            X_tr = X_subset.iloc[train_idx]\n            y_tr = y_eval.iloc[train_idx]\n            X_val = X_subset.iloc[val_idx]\n            y_val = y_eval.iloc[val_idx]\n            \n            model.fit(X_tr, y_tr)\n            pred = model.predict(X_val)\n            score = pearsonr(y_val, pred)[0]\n            scores.append(score)\n        \n        return np.mean(scores), np.std(scores)\n    \n    def save_results(self, output_path='transformer_feature_selection_results.json'):\n        \"\"\"Save results.\"\"\"\n        serializable_results = {}\n        for key, value in self.selection_results.items():\n            if isinstance(value, (np.integer, np.floating)):\n                serializable_results[key] = float(value)\n            elif isinstance(value, np.ndarray):\n                serializable_results[key] = value.tolist()\n            elif isinstance(value, dict):\n                serializable_dict = {}\n                for k, v in value.items():\n                    if isinstance(v, dict):\n                        serializable_dict[k] = {\n                            sub_k: float(sub_v) if isinstance(sub_v, (np.integer, np.floating)) else sub_v\n                            for sub_k, sub_v in v.items()\n                        }\n                    elif isinstance(v, (np.integer, np.floating)):\n                        serializable_dict[k] = float(v)\n                    else:\n                        serializable_dict[k] = v\n                serializable_results[key] = serializable_dict\n            else:\n                serializable_results[key] = value\n        \n        with open(output_path, 'w') as f:\n            json.dump(serializable_results, f, indent=2)\n        \n        print(f\"\\nResults saved to {output_path}\")\n\ndef run_memory_optimized_transformer_selection(\n    train_path=\"/kaggle/input/drw-crypto-market-prediction/train.parquet\",\n    test_path=\"/kaggle/input/drw-crypto-market-prediction/test.parquet\",\n    initial_sample_size=20000,\n    prescreening_features=150,\n    model_sample_size=5000,\n    n_transformer_models=2\n):\n    \"\"\"Execute memory-optimized transformer feature selection.\"\"\"\n    \n    config = {\n        'train_path': train_path,\n        'test_path': test_path,\n        'initial_sample_size': initial_sample_size,\n        'prescreening_features': prescreening_features,\n        'model_sample_size': model_sample_size,\n        'n_transformer_models': n_transformer_models\n    }\n    \n    selector = MemoryOptimizedTransformerSelector(config)\n    \n    print(\"Starting memory-optimized transformer feature selection...\")\n    print(f\"Initial sample: {initial_sample_size:,} rows\")\n    print(f\"Pre-screening to: {prescreening_features} features\")\n    print(f\"Model training sample: {model_sample_size:,} rows\")\n    print(f\"Transformer models: {n_transformer_models}\")\n    selector.log_memory(\"Initial state\")\n    \n    # Load data sample\n    print(f\"\\nLoading {initial_sample_size} rows...\")\n    \n    # Get total rows\n    train_df_info = pd.read_parquet(train_path, columns=['label'])\n    total_rows = len(train_df_info)\n    print(f\"Total dataset size: {total_rows} rows\")\n    del train_df_info\n    gc.collect()\n    \n    # Sample focusing on recent data\n    np.random.seed(42)\n    weights = np.linspace(0.7, 1.0, total_rows)\n    weights = weights / weights.sum()\n    sample_indices = np.random.choice(total_rows, size=min(initial_sample_size, total_rows), \n                                     replace=False, p=weights)\n    sample_indices.sort()\n    \n    # Load sampled data\n    train_df = pd.read_parquet(train_path).iloc[sample_indices].reset_index(drop=True)\n    \n    # Get features\n    feature_cols = [col for col in train_df.columns if col not in [\"timestamp\", \"label\"]]\n    baseline_features = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    \n    print(f\"Initial features: {len(feature_cols)}\")\n    \n    # Separate features and target\n    X_train = train_df[feature_cols].copy()\n    y_train = train_df[\"label\"].copy()\n    \n    del train_df\n    gc.collect()\n    \n    # Data preprocessing\n    print(\"\\nOptimizing data types...\")\n    X_train = selector.reduce_data_precision(X_train)\n    X_train = selector.validate_and_clean_data(X_train)\n    \n    # Aggressive variance filtering\n    print(\"\\nRemoving low-variance features...\")\n    X_train_filtered, high_variance_features, removed_features = selector.remove_low_variance_features(\n        X_train, threshold_percentile=30  # More aggressive\n    )\n    \n    # Ensure baseline features\n    for feature in baseline_features:\n        if feature in removed_features and feature in X_train.columns:\n            X_train_filtered[feature] = X_train[feature]\n            high_variance_features.append(feature)\n    \n    del X_train\n    gc.collect()\n    \n    # Aggressive pre-screening with mutual information\n    if len(high_variance_features) > prescreening_features:\n        X_train_prescreened, prescreened_features = selector.aggressive_feature_prescreening(\n            X_train_filtered, y_train, target_features=prescreening_features\n        )\n        \n        # Ensure baseline features in prescreened set\n        for feature in baseline_features:\n            if feature not in prescreened_features and feature in high_variance_features:\n                prescreened_features.append(feature)\n                X_train_prescreened[feature] = X_train_filtered[feature]\n    else:\n        X_train_prescreened = X_train_filtered\n        prescreened_features = high_variance_features\n    \n    del X_train_filtered\n    gc.collect()\n    selector.log_memory(\"After prescreening\")\n    \n    # Run transformer selection\n    print(\"\\nTraining lightweight transformer ensemble...\")\n    start_time = time.time()\n    \n    aggregated_importance = selector.ensemble_transformer_selection(\n        X_train_prescreened, y_train, prescreened_features,\n        n_models=n_transformer_models,\n        sample_size=model_sample_size\n    )\n    \n    # Sort features\n    sorted_features = sorted(\n        aggregated_importance.items(), \n        key=lambda x: x[1]['mean'], \n        reverse=True\n    )\n    \n    # Evaluate different feature counts\n    print(\"\\nEvaluating feature sets...\")\n    \n    eval_sample_size = min(10000, len(X_train_prescreened))\n    eval_indices = np.random.choice(len(X_train_prescreened), size=eval_sample_size, replace=False)\n    X_eval = X_train_prescreened.iloc[eval_indices].copy()\n    y_eval = y_train.iloc[eval_indices].copy()\n    \n    # Test different feature counts\n    feature_counts = [20, 30, 40, 50, 75]\n    best_score = -1\n    best_n_features = None\n    best_features = []\n    \n    evaluation_results = {}\n    \n    for n_features in feature_counts:\n        if n_features > len(sorted_features):\n            continue\n            \n        selected_features = [f[0] for f in sorted_features[:n_features]]\n        \n        # Ensure baseline features\n        for feature in baseline_features:\n            if feature not in selected_features and feature in prescreened_features:\n                selected_features.append(feature)\n        \n        print(f\"\\nEvaluating top {len(selected_features)} features...\")\n        \n        try:\n            cv_score, cv_std = selector.evaluate_feature_set(X_eval, y_eval, selected_features)\n            \n            evaluation_results[f'top_{n_features}'] = {\n                'n_features': len(selected_features),\n                'cv_score': float(cv_score),\n                'cv_std': float(cv_std)\n            }\n            \n            print(f\"  CV score: {cv_score:.6f} (±{cv_std:.6f})\")\n            \n            if cv_score > best_score:\n                best_score = cv_score\n                best_n_features = n_features\n                best_features = selected_features\n        except Exception as e:\n            print(f\"  Error: {str(e)}\")\n            continue\n    \n    # Store results\n    selector.selection_results = {\n        'aggregated_importance': {k: v for k, v in sorted_features[:100]},\n        'evaluation_results': evaluation_results,\n        'best_n_features': best_n_features,\n        'best_features': best_features,\n        'best_score': float(best_score) if best_score > -1 else None,\n        'runtime_minutes': (time.time() - start_time) / 60,\n        'config': config\n    }\n    \n    # Save results\n    selector.save_results()\n    \n    # Print summary\n    print(\"\\n\" + \"=\"*60)\n    print(\"MEMORY-OPTIMIZED TRANSFORMER FEATURE SELECTION SUMMARY\")\n    print(\"=\"*60)\n    \n    print(f\"\\nTop 20 features by transformer attention importance:\")\n    for i, (feature, scores) in enumerate(sorted_features[:20]):\n        print(f\"{i+1:2d}. {feature:10s} - Mean: {scores['mean']:.4f}\")\n    \n    print(f\"\\nBest configuration: Top {best_n_features} features\")\n    print(f\"Best CV score: {best_score:.6f}\")\n    print(f\"Runtime: {(time.time() - start_time) / 60:.1f} minutes\")\n    \n    print(\"\\nBaseline feature importance:\")\n    for feature in baseline_features:\n        if feature in aggregated_importance:\n            scores = aggregated_importance[feature]\n            print(f\"  {feature}: {scores['mean']:.4f}\")\n    \n    return selector, best_features\n\ndef train_final_model_and_generate_submission(\n    selected_features,\n    train_path=\"/kaggle/input/drw-crypto-market-prediction/train.parquet\",\n    test_path=\"/kaggle/input/drw-crypto-market-prediction/test.parquet\",\n    submission_path=\"submission.csv\",\n    train_sample_size=100000,\n    use_ensemble=True\n):\n    \"\"\"\n    Train final model on selected features and generate competition submission.\n    \n    Parameters:\n    -----------\n    selected_features : list\n        List of selected feature names\n    train_path : str\n        Path to training data\n    test_path : str\n        Path to test data\n    submission_path : str\n        Path for output submission file\n    train_sample_size : int\n        Number of training samples to use (None for all data)\n    use_ensemble : bool\n        Whether to use ensemble of models\n    \"\"\"\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"TRAINING FINAL MODEL AND GENERATING SUBMISSION\")\n    print(\"=\"*60)\n    \n    # Load training data\n    print(f\"\\nLoading training data for final model...\")\n    if train_sample_size is not None:\n        # Get total rows for sampling\n        train_df_info = pd.read_parquet(train_path, columns=['label'])\n        total_rows = len(train_df_info)\n        del train_df_info\n        gc.collect()\n        \n        # Sample with preference for recent data\n        np.random.seed(42)\n        weights = np.linspace(0.5, 1.0, total_rows)\n        weights = weights / weights.sum()\n        sample_indices = np.random.choice(total_rows, size=min(train_sample_size, total_rows), \n                                         replace=False, p=weights)\n        sample_indices.sort()\n        \n        train_df = pd.read_parquet(train_path).iloc[sample_indices].reset_index(drop=True)\n        print(f\"Using {len(train_df):,} training samples\")\n    else:\n        train_df = pd.read_parquet(train_path)\n        print(f\"Using full training data: {len(train_df):,} samples\")\n    \n    # Prepare training data\n    X_train = train_df[selected_features]\n    y_train = train_df['label']\n    \n    # Data validation and cleaning\n    print(\"Validating training data...\")\n    # Handle any remaining data quality issues\n    X_train = X_train.replace([np.inf, -np.inf], np.nan)\n    X_train = X_train.fillna(X_train.median())\n    \n    # Convert to float32 for memory efficiency\n    X_train = X_train.astype(np.float32)\n    y_train = y_train.astype(np.float32)\n    \n    del train_df\n    gc.collect()\n    \n    # Load test data\n    print(\"\\nLoading test data...\")\n    test_df = pd.read_parquet(test_path)\n    test_ids = test_df.index  # Preserve order for submission\n    X_test = test_df[selected_features]\n    \n    # Validate test data\n    print(\"Validating test data...\")\n    X_test = X_test.replace([np.inf, -np.inf], np.nan)\n    X_test = X_test.fillna(X_test.median())\n    X_test = X_test.astype(np.float32)\n    \n    del test_df\n    gc.collect()\n    \n    # Scale features\n    print(\"\\nScaling features...\")\n    scaler = RobustScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_test_scaled = scaler.transform(X_test)\n    \n    # Train model(s)\n    if use_ensemble:\n        print(\"\\nTraining XGBoost ensemble...\")\n        predictions = np.zeros(len(X_test))\n        n_models = 5\n        \n        for i in range(n_models):\n            print(f\"Training model {i+1}/{n_models}...\")\n            \n            # Different random state for each model\n            model = XGBRegressor(\n                n_estimators=300,\n                max_depth=6,\n                learning_rate=0.05,\n                subsample=0.8,\n                colsample_bytree=0.8,\n                random_state=42 + i,\n                n_jobs=-1,\n                tree_method='hist',\n                objective='reg:squarederror'\n            )\n            \n            # Train on scaled data\n            model.fit(X_train_scaled, y_train)\n            \n            # Add predictions\n            predictions += model.predict(X_test_scaled) / n_models\n            \n            # Feature importance from best model\n            if i == 0:\n                feature_importance = pd.DataFrame({\n                    'feature': selected_features,\n                    'importance': model.feature_importances_\n                }).sort_values('importance', ascending=False)\n                \n                print(\"\\nTop 10 most important features in final model:\")\n                for idx, row in feature_importance.head(10).iterrows():\n                    print(f\"  {row['feature']}: {row['importance']:.4f}\")\n            \n            gc.collect()\n    \n    else:\n        print(\"\\nTraining single XGBoost model...\")\n        model = XGBRegressor(\n            n_estimators=500,\n            max_depth=6,\n            learning_rate=0.03,\n            subsample=0.8,\n            colsample_bytree=0.8,\n            random_state=42,\n            n_jobs=-1,\n            tree_method='hist',\n            objective='reg:squarederror'\n        )\n        \n        model.fit(X_train_scaled, y_train)\n        predictions = model.predict(X_test_scaled)\n        \n        # Feature importance\n        feature_importance = pd.DataFrame({\n            'feature': selected_features,\n            'importance': model.feature_importances_\n        }).sort_values('importance', ascending=False)\n        \n        print(\"\\nTop 10 most important features in final model:\")\n        for idx, row in feature_importance.head(10).iterrows():\n            print(f\"  {row['feature']}: {row['importance']:.4f}\")\n    \n    # Create submission with correct column name\n    print(\"\\nCreating submission file...\")\n    submission = pd.DataFrame({\n        'id': test_ids,\n        'prediction': predictions  # Changed from 'label' to 'prediction'\n    })\n    \n    # Save submission\n    submission.to_csv(submission_path, index=False)\n    print(f\"Submission saved to: {submission_path}\")\n    \n    # Print submission statistics\n    print(\"\\nSubmission statistics:\")\n    print(f\"  Number of predictions: {len(submission):,}\")\n    print(f\"  Prediction mean: {predictions.mean():.6f}\")\n    print(f\"  Prediction std: {predictions.std():.6f}\")\n    print(f\"  Prediction min: {predictions.min():.6f}\")\n    print(f\"  Prediction max: {predictions.max():.6f}\")\n    \n    # Validate submission format\n    print(\"\\nValidating submission format...\")\n    assert len(submission) == len(X_test), \"Submission length mismatch\"\n    assert submission['prediction'].isna().sum() == 0, \"Submission contains NaN values\"\n    assert 'id' in submission.columns, \"Missing 'id' column\"\n    assert 'prediction' in submission.columns, \"Missing 'prediction' column\"\n    print(\"Submission validation passed ✓\")\n    \n    return submission, feature_importance\n\n\nif __name__ == \"__main__\":\n    # Run with very conservative parameters\n    selector, selected_features = run_memory_optimized_transformer_selection(\n        initial_sample_size=30000,      # Smaller initial sample\n        prescreening_features=250,       # Aggressive prescreening\n        model_sample_size=50000,          # Very small transformer training samples\n        n_transformer_models=2           # Minimal ensemble\n    )\n    \n    print(f\"\\nFinal selection: {len(selected_features)} features\")\n    print(\"\\nSelected features:\")\n    for i, feature in enumerate(selected_features):\n        print(f\"  {i+1}. {feature}\")\n    \n    # Train final model and generate submission\n    submission, feature_importance = train_final_model_and_generate_submission(\n        selected_features=selected_features,\n        train_sample_size=100000,  # Use 100k samples for final training\n        use_ensemble=True          # Use ensemble for better performance\n    )\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"COMPETITION SUBMISSION COMPLETE\")\n    print(\"=\"*60)\n    print(f\"Selected features: {len(selected_features)}\")\n    print(f\"Submission file: submission.csv\")\n    print(f\"Ready for upload to Kaggle competition\")\n    \n    # Save feature list for reference\n    with open('selected_features.txt', 'w') as f:\n        for feature in selected_features:\n            f.write(f\"{feature}\\n\")\n    print(\"\\nSelected features saved to: selected_features.txt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:44:46.516605Z","iopub.execute_input":"2025-05-28T20:44:46.51729Z","iopub.status.idle":"2025-05-28T20:45:29.481637Z","shell.execute_reply.started":"2025-05-28T20:44:46.517262Z","shell.execute_reply":"2025-05-28T20:45:29.477685Z"}},"outputs":[],"execution_count":null}]}