{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nTANGOS Feature Importance Analysis and Prediction for DRW Crypto Market Prediction\nEnhanced version with incremental improvements:\n- Advanced feature engineering\n- Improved model architecture with residual connections\n- Better training strategy (cosine annealing, gradient accumulation)\n- Ensemble predictions\n- Post-processing\n\"\"\"\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import spearmanr, pearsonr, rankdata\nfrom scipy.ndimage import gaussian_filter1d\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport os\nimport sys\nimport gc\nimport random\nfrom typing import List, Tuple, Dict\n\nwarnings.filterwarnings(\"ignore\")\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\ntorch.manual_seed(42)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed(42)\n\n# Set style for visualizations\nplt.style.use('seaborn-v0_8-darkgrid')\nsns.set_palette(\"husl\")\n\n# Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Feature Engineering\n# =========================\ndef add_features(df):\n    \"\"\"Add engineered features to the dataframe\"\"\"\n    # Store original columns to avoid modifying them\n    original_cols = df.columns.tolist()\n    \n    # Basic interactions\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    \n    # Volume-based features\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n    \n    # Market microstructure features\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # Order flow features\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    \n    # Depth features\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Market impact proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Activity indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    \n    # Log transformations\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Volatility proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex interactions\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market efficiency\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market stress\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # === NEW FEATURES ===\n    # Percentile features for better normalization\n    for col in ['volume', 'total_depth', 'net_order_flow', 'bid_qty', 'ask_qty']:\n        if col in df.columns:\n            df[f'{col}_percentile'] = df[col].rank(pct=True)\n    \n    # Advanced ratio features\n    df['bid_ask_volume_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    df['trade_imbalance_ratio'] = df['order_flow_imbalance'] * df['depth_imbalance']\n    \n    # Volume-depth logarithmic ratios\n    df['volume_depth_log_ratio'] = np.log1p(df['volume']) / (np.log1p(df['total_depth']) + 1e-10)\n    df['sqrt_flow_volume'] = np.sqrt(np.abs(df['net_order_flow'])) * np.sign(df['net_order_flow'])\n    \n    # Harmonic means\n    df['harmonic_mean_depth'] = 2 / (1/(df['bid_qty'] + 1e-10) + 1/(df['ask_qty'] + 1e-10))\n    df['harmonic_mean_trade'] = 2 / (1/(df['buy_qty'] + 1e-10) + 1/(df['sell_qty'] + 1e-10))\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df\n\n# =========================\n# Model Components\n# =========================\nclass StableOrthogonalProjection(nn.Module):\n    \"\"\"Stable orthogonal projection using QR decomposition\"\"\"\n    def __init__(self, input_dim, output_dim):\n        super().__init__()\n        self.input_dim = input_dim\n        self.output_dim = min(output_dim, input_dim)\n        \n        self.weight = nn.Parameter(torch.empty(self.output_dim, input_dim))\n        nn.init.orthogonal_(self.weight)\n        \n        self.bias = nn.Parameter(torch.zeros(self.output_dim))\n        \n    def forward(self, x):\n        Q, R = torch.linalg.qr(self.weight.t())\n        orthogonal_weight = Q[:, :self.output_dim].t()\n        \n        return F.linear(x, orthogonal_weight, self.bias)\n\nclass ResidualBlock(nn.Module):\n    \"\"\"Residual block with batch normalization\"\"\"\n    def __init__(self, dim, dropout_rate=0.3):\n        super().__init__()\n        self.linear = nn.Linear(dim, dim)\n        self.bn = nn.BatchNorm1d(dim)\n        self.dropout = nn.Dropout(dropout_rate)\n        self.activation = nn.ReLU()\n        \n    def forward(self, x):\n        residual = x\n        out = self.linear(x)\n        out = self.bn(out)\n        out = self.activation(out)\n        out = self.dropout(out)\n        return out + residual\n\nclass FeatureSelector(nn.Module):\n    \"\"\"Feature selection with learnable gates\"\"\"\n    def __init__(self, input_dim, temperature=1.0):\n        super().__init__()\n        self.input_dim = input_dim\n        self.temperature = temperature\n        \n        self.importance_logits = nn.Parameter(torch.randn(input_dim) * 0.01)\n        \n    def forward(self, x, hard=False):\n        gates = torch.sigmoid(self.importance_logits / self.temperature)\n        \n        if hard:\n            gates = (gates > 0.5).float()\n        \n        selected_features = x * gates.unsqueeze(0)\n        \n        return selected_features, gates\n    \n    def get_feature_ranking(self):\n        with torch.no_grad():\n            scores = torch.sigmoid(self.importance_logits)\n            ranking = torch.argsort(scores, descending=True)\n            return ranking, scores\n\nclass ImprovedTANGOSEncoder(nn.Module):\n    \"\"\"Enhanced encoder with residual connections\"\"\"\n    def __init__(self, input_dim, hidden_dims=[256, 128, 64], dropout_rate=0.3, use_residual=True):\n        super().__init__()\n        \n        self.use_residual = use_residual\n        layers = []\n        prev_dim = input_dim\n        \n        for i, hidden_dim in enumerate(hidden_dims):\n            if i < len(hidden_dims) - 1:\n                hidden_dim = min(hidden_dim, prev_dim)\n                layers.append(StableOrthogonalProjection(prev_dim, hidden_dim))\n            else:\n                layers.append(nn.Linear(prev_dim, hidden_dim))\n            \n            layers.append(nn.BatchNorm1d(hidden_dim))\n            layers.append(nn.ReLU())\n            layers.append(nn.Dropout(dropout_rate))\n            \n            # Add residual block every other layer\n            if use_residual and i % 2 == 1 and i < len(hidden_dims) - 1:\n                layers.append(ResidualBlock(hidden_dim, dropout_rate))\n            \n            prev_dim = hidden_dim\n        \n        self.layers = nn.ModuleList(layers)\n        self.output_dim = prev_dim\n        \n    def forward(self, x):\n        for layer in self.layers:\n            x = layer(x)\n        return x\n\nclass ImprovedTANGOS(nn.Module):\n    \"\"\"Enhanced TANGOS with residual connections\"\"\"\n    def __init__(self, input_dim, encoder_dims=[256, 128, 64], \n                 predictor_dims=[32, 16], dropout_rate=0.3, temperature=1.0, use_residual=True):\n        super().__init__()\n        \n        self.input_dim = input_dim\n        \n        self.feature_selector = FeatureSelector(input_dim, temperature)\n        self.encoder = ImprovedTANGOSEncoder(input_dim, encoder_dims, dropout_rate, use_residual)\n        \n        layers = []\n        prev_dim = self.encoder.output_dim\n        \n        for hidden_dim in predictor_dims:\n            layers.append(nn.Linear(prev_dim, hidden_dim))\n            layers.append(nn.BatchNorm1d(hidden_dim))\n            layers.append(nn.ReLU())\n            layers.append(nn.Dropout(dropout_rate))\n            prev_dim = hidden_dim\n        \n        layers.append(nn.Linear(prev_dim, 1))\n        self.predictor = nn.Sequential(*layers)\n        \n        self.register_buffer('gradient_importance', torch.zeros(input_dim))\n        self.register_buffer('gradient_count', torch.tensor(0))\n        \n    def forward(self, x, track_gradients=False):\n        selected_x, gates = self.feature_selector(x)\n        encoded = self.encoder(selected_x)\n        output = self.predictor(encoded)\n        \n        if track_gradients and x.requires_grad:\n            self.track_feature_gradients(x, output)\n        \n        return output, gates\n    \n    def track_feature_gradients(self, x, output):\n        if output.requires_grad:\n            grad = torch.autograd.grad(\n                outputs=output.sum(),\n                inputs=x,\n                retain_graph=True,\n                create_graph=False\n            )[0]\n            \n            self.gradient_importance += torch.abs(grad).mean(dim=0).detach()\n            self.gradient_count += 1\n    \n    def get_gradient_importance(self):\n        if self.gradient_count > 0:\n            return self.gradient_importance / self.gradient_count\n        return torch.zeros(self.input_dim, device=self.gradient_importance.device)\n\n# =========================\n# Training and Evaluation\n# =========================\ndef train_model_improved(model, train_loader, val_loader, n_epochs=20, patience=5, \n                        device='cuda', accumulation_steps=2):\n    \"\"\"Enhanced training with cosine annealing and gradient accumulation\"\"\"\n    criterion = nn.HuberLoss(delta=1.0)\n    optimizer = torch.optim.AdamW(model.parameters(), lr=0.003, weight_decay=1e-4)\n    \n    # Cosine annealing with warm restarts\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingWarmRestarts(\n        optimizer, T_0=5, T_mult=2, eta_min=1e-5\n    )\n    \n    best_val_loss = float('inf')\n    best_val_corr = -float('inf')\n    patience_counter = 0\n    history = {'train_loss': [], 'val_loss': [], 'val_corr': []}\n    \n    for epoch in range(n_epochs):\n        # Training\n        model.train()\n        train_loss = 0.0\n        \n        progress_bar = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{n_epochs}\")\n        for i, (inputs, targets) in enumerate(progress_bar):\n            inputs = inputs.to(device).requires_grad_(True)\n            targets = targets.to(device)\n            \n            outputs, gates = model(inputs, track_gradients=True)\n            loss = criterion(outputs, targets)\n            \n            sparsity_loss = 0.01 * gates.mean()\n            total_loss = (loss + sparsity_loss) / accumulation_steps\n            \n            total_loss.backward()\n            \n            if (i + 1) % accumulation_steps == 0:\n                torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n                optimizer.step()\n                optimizer.zero_grad()\n            \n            train_loss += loss.item()\n            progress_bar.set_postfix({'loss': loss.item()})\n        \n        # Step scheduler\n        scheduler.step()\n        \n        # Validation\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        \n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs = inputs.to(device)\n                targets = targets.to(device)\n                \n                outputs, _ = model(inputs)\n                loss = criterion(outputs, targets)\n                val_loss += loss.item()\n                \n                val_preds.extend(outputs.cpu().numpy().flatten())\n                val_targets.extend(targets.cpu().numpy().flatten())\n        \n        avg_train_loss = train_loss / len(train_loader)\n        avg_val_loss = val_loss / len(val_loader)\n        val_corr = pearsonr(val_targets, val_preds)[0]\n        \n        history['train_loss'].append(avg_train_loss)\n        history['val_loss'].append(avg_val_loss)\n        history['val_corr'].append(val_corr)\n        \n        print(f\"Epoch {epoch+1}: Train Loss={avg_train_loss:.4f}, \"\n              f\"Val Loss={avg_val_loss:.4f}, Val Corr={val_corr:.4f}, \"\n              f\"LR={optimizer.param_groups[0]['lr']:.6f}\")\n        \n        if val_corr > best_val_corr:\n            best_val_corr = val_corr\n            best_val_loss = avg_val_loss\n            torch.save(model.state_dict(), 'best_tangos_model.pt')\n            patience_counter = 0\n        else:\n            patience_counter += 1\n        \n        if patience_counter >= patience:\n            print(f\"Early stopping triggered at epoch {epoch+1}\")\n            break\n    \n    model.load_state_dict(torch.load('best_tangos_model.pt'))\n    \n    return model, history\n\ndef compute_feature_importance_fast(model, all_features):\n    \"\"\"Fast feature importance using only selection scores and gradients\"\"\"\n    model.eval()\n    \n    _, selection_scores = model.feature_selector.get_feature_ranking()\n    selection_scores = selection_scores.cpu().numpy()\n    gradient_importance = model.get_gradient_importance().cpu().numpy()\n    \n    # Normalize scores\n    def normalize_scores(scores):\n        if scores.max() > scores.min():\n            return (scores - scores.min()) / (scores.max() - scores.min())\n        return np.zeros_like(scores)\n    \n    selection_norm = normalize_scores(selection_scores)\n    gradient_norm = normalize_scores(gradient_importance)\n    \n    # Combined importance\n    combined_importance = 0.6 * selection_norm + 0.4 * gradient_norm\n    \n    results_df = pd.DataFrame({\n        'feature': all_features,\n        'combined_importance': combined_importance,\n        'selection_score': selection_scores,\n        'gradient_importance': gradient_importance\n    })\n    \n    results_df = results_df.sort_values('combined_importance', ascending=False)\n    \n    return results_df\n\ndef create_ensemble_predictions(train_df, test_df, top_features, n_models=3, device='cuda'):\n    \"\"\"Train ensemble of models with different configurations\"\"\"\n    predictions_list = []\n    \n    # Different model configurations\n    configs = [\n        {'encoder_dims': [256, 128, 64], 'predictor_dims': [32, 16], 'dropout': 0.2, 'lr': 0.003},\n        {'encoder_dims': [128, 64, 32], 'predictor_dims': [16, 8], 'dropout': 0.25, 'lr': 0.002},\n        {'encoder_dims': [512, 256, 128], 'predictor_dims': [64, 32], 'dropout': 0.15, 'lr': 0.004}\n    ]\n    \n    for i in range(n_models):\n        print(f\"\\n🔄 Training ensemble model {i+1}/{n_models}\")\n        \n        # Set different random seed\n        seed = 42 + i\n        np.random.seed(seed)\n        torch.manual_seed(seed)\n        \n        # Get config\n        config = configs[i % len(configs)]\n        \n        # Prepare data\n        X_full = train_df[top_features].values\n        y_full = train_df['label'].values\n        \n        # Train/validation split with different seed\n        X_train, X_val, y_train, y_val = train_test_split(\n            X_full, y_full, test_size=0.15, random_state=seed\n        )\n        \n        # Scale features with RobustScaler for variety\n        if i % 2 == 0:\n            scaler = StandardScaler()\n        else:\n            scaler = RobustScaler()\n            \n        X_train_scaled = scaler.fit_transform(X_train)\n        X_val_scaled = scaler.transform(X_val)\n        \n        # Create data loaders\n        batch_size = 1024 if device.type == 'cuda' else 256\n        \n        train_dataset = TensorDataset(\n            torch.tensor(X_train_scaled, dtype=torch.float32),\n            torch.tensor(y_train, dtype=torch.float32).unsqueeze(1)\n        )\n        val_dataset = TensorDataset(\n            torch.tensor(X_val_scaled, dtype=torch.float32),\n            torch.tensor(y_val, dtype=torch.float32).unsqueeze(1)\n        )\n        \n        train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\n        val_loader = DataLoader(val_dataset, batch_size=batch_size*2, shuffle=False)\n        \n        # Initialize model\n        model = ImprovedTANGOS(\n            input_dim=len(top_features),\n            encoder_dims=config['encoder_dims'],\n            predictor_dims=config['predictor_dims'],\n            dropout_rate=config['dropout'],\n            temperature=2.0,\n            use_residual=True\n        ).to(device)\n        \n        # Train model\n        model, _ = train_model_improved(\n            model, train_loader, val_loader, \n            n_epochs=15, patience=5, device=device\n        )\n        \n        # Generate predictions\n        test_features = test_df[top_features].values\n        test_features_scaled = scaler.transform(test_features)\n        test_tensor = torch.tensor(test_features_scaled, dtype=torch.float32, device=device)\n        \n        model.eval()\n        predictions = []\n        \n        with torch.no_grad():\n            for j in range(0, len(test_tensor), batch_size * 4):\n                batch = test_tensor[j:j+batch_size * 4]\n                batch_preds, _ = model(batch)\n                predictions.extend(batch_preds.cpu().numpy().flatten())\n        \n        predictions_list.append(np.array(predictions))\n        \n        # Clean up\n        del model\n        gc.collect()\n        if device.type == 'cuda':\n            torch.cuda.empty_cache()\n    \n    return predictions_list\n\ndef post_process_predictions(predictions, method='clip_smooth'):\n    \"\"\"Post-process predictions for better results\"\"\"\n    \n    if method == 'clip_smooth':\n        # Clip extreme values\n        p5, p95 = np.percentile(predictions, [5, 95])\n        predictions = np.clip(predictions, p5, p95)\n        \n        # Smooth predictions slightly\n        predictions = gaussian_filter1d(predictions, sigma=1.0)\n        \n    elif method == 'rank_transform':\n        # Rank transformation\n        predictions = rankdata(predictions) / len(predictions)\n        \n    return predictions\n\n# =========================\n# Main Function\n# =========================\ndef main():\n    print(\"=== Enhanced TANGOS for DRW Competition ===\")\n    print(\"=== With ensemble and improved training ===\\n\")\n    \n    # Data paths\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_submission_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # Load data\n    print(\"Loading competition data...\")\n    \n    # Define features\n    x_features = [f\"X{i}\" for i in range(1, 891)]\n    market_features = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    \n    # Load training data\n    train_df = pd.read_parquet(train_path)\n    print(f\"Loaded {len(train_df)} training samples\")\n    \n    # Load test data\n    test_df = pd.read_parquet(test_path)\n    print(f\"Loaded {len(test_df)} test samples\")\n    \n    # Load sample submission\n    sample_submission = pd.read_csv(sample_submission_path)\n    print(f\"Sample submission shape: {sample_submission.shape}\")\n    \n    # IMPORTANT: IDs must start from 1, not 0\n    test_ids = np.arange(1, len(test_df) + 1)\n    print(f\"Test IDs range: {test_ids[0]} to {test_ids[-1]}\")\n    \n    # Add engineered features\n    print(\"\\nAdding engineered features...\")\n    train_df = add_features(train_df)\n    test_df = add_features(test_df)\n    \n    # Garbage collection\n    gc.collect()\n    \n    # Get all feature names\n    engineered_features = [col for col in train_df.columns \n                          if col not in x_features + market_features + ['timestamp', 'label']]\n    all_features = x_features + market_features + engineered_features\n    \n    print(f\"Total features: {len(all_features)}\")\n    print(f\"- Anonymous features (X_): {len(x_features)}\")\n    print(f\"- Market features: {len(market_features)}\")\n    print(f\"- Engineered features: {len(engineered_features)}\")\n    \n    # Create outputs directory\n    os.makedirs('outputs', exist_ok=True)\n    \n    # Feature importance analysis\n    print(\"\\n\" + \"=\"*60)\n    print(\"FEATURE IMPORTANCE ANALYSIS\")\n    print(\"=\"*60)\n    \n    # Use 50k samples for feature importance analysis\n    feature_sample_size = 50000\n    train_sample = train_df.sample(n=min(feature_sample_size, len(train_df)), random_state=42)\n    \n    X_feat = train_sample[all_features].values\n    y_feat = train_sample['label'].values\n    \n    # Train/validation split\n    X_train_feat, X_val_feat, y_train_feat, y_val_feat = train_test_split(\n        X_feat, y_feat, test_size=0.2, random_state=42\n    )\n    \n    # Scale features\n    scaler_feat = StandardScaler()\n    X_train_feat_scaled = scaler_feat.fit_transform(X_train_feat)\n    X_val_feat_scaled = scaler_feat.transform(X_val_feat)\n    \n    # Create data loaders\n    batch_size = 256 if device.type == 'cpu' else 1024\n    \n    train_dataset_feat = TensorDataset(\n        torch.tensor(X_train_feat_scaled, dtype=torch.float32),\n        torch.tensor(y_train_feat, dtype=torch.float32).unsqueeze(1)\n    )\n    val_dataset_feat = TensorDataset(\n        torch.tensor(X_val_feat_scaled, dtype=torch.float32),\n        torch.tensor(y_val_feat, dtype=torch.float32).unsqueeze(1)\n    )\n    \n    train_loader_feat = DataLoader(train_dataset_feat, batch_size=batch_size, shuffle=True)\n    val_loader_feat = DataLoader(val_dataset_feat, batch_size=batch_size*2, shuffle=False)\n    \n    # Initialize and train model for feature importance\n    print(\"\\nTraining model for feature importance...\")\n    model_feat = ImprovedTANGOS(\n        input_dim=len(all_features),\n        encoder_dims=[512, 256, 128],\n        predictor_dims=[64, 32],\n        dropout_rate=0.3,\n        temperature=1.0,\n        use_residual=True\n    ).to(device)\n    \n    n_epochs = 5 if device.type == 'cpu' else 10\n    model_feat, _ = train_model_improved(model_feat, train_loader_feat, val_loader_feat, \n                                        n_epochs=n_epochs, device=device)\n    \n    # Compute feature importance\n    print(\"\\nComputing feature importance...\")\n    results_df = compute_feature_importance_fast(model_feat, all_features)\n    \n    # Try different numbers of top features\n    feature_counts = [80, 100]  # Test with 80 and 100 features\n    \n    for TOP_K_FEATURES in feature_counts:\n        print(f\"\\n{'='*60}\")\n        print(f\"CREATING SUBMISSION WITH TOP {TOP_K_FEATURES} FEATURES\")\n        print(f\"{'='*60}\")\n        \n        top_features = results_df.head(TOP_K_FEATURES)['feature'].tolist()\n        \n        print(f\"\\n🏆 Selected top {TOP_K_FEATURES} features:\")\n        print(\"-\" * 60)\n        for i in range(min(20, TOP_K_FEATURES)):\n            row = results_df.iloc[i]\n            print(f\"{i+1:3d}. {row['feature']:25s} {row['combined_importance']:.6f}\")\n        if TOP_K_FEATURES > 20:\n            print(f\"... ({TOP_K_FEATURES - 20} more features)\")\n        \n        # Save feature importance\n        results_df.to_csv(f'outputs/tangos_feature_importance_{TOP_K_FEATURES}.csv', index=False)\n        \n        # Save top features list\n        with open(f'outputs/top_{TOP_K_FEATURES}_features.txt', 'w') as f:\n            for i, feat in enumerate(top_features, 1):\n                f.write(f\"{i}. {feat}\\n\")\n        \n        # Clean up\n        if feature_counts.index(TOP_K_FEATURES) == 0:\n            del model_feat, X_train_feat_scaled, X_val_feat_scaled\n            gc.collect()\n            if device.type == 'cuda':\n                torch.cuda.empty_cache()\n        \n        # Create ensemble predictions\n        print(f\"\\nCreating ensemble predictions with {TOP_K_FEATURES} features...\")\n        ensemble_predictions = create_ensemble_predictions(\n            train_df, test_df, top_features, n_models=3, device=device\n        )\n        \n        # Combine ensemble predictions\n        # Try different ensemble strategies\n        ensemble_strategies = {\n            'mean': np.mean(ensemble_predictions, axis=0),\n            'median': np.median(ensemble_predictions, axis=0),\n            'weighted': np.average(ensemble_predictions, axis=0, \n                                  weights=[0.4, 0.35, 0.25])\n        }\n        \n        for strategy_name, predictions in ensemble_strategies.items():\n            # Post-process predictions\n            predictions_processed = post_process_predictions(predictions)\n            \n            # Create submission\n            submission = pd.DataFrame({\n                'ID': test_ids,\n                'prediction': predictions_processed\n            })\n            \n            # Save submission\n            filename = f'submission_{TOP_K_FEATURES}features_{strategy_name}.csv'\n            submission.to_csv(filename, index=False)\n            print(f\"✅ Saved: {filename}\")\n            \n            # Print stats\n            print(f\"   Mean: {submission['prediction'].mean():.6f}, \"\n                  f\"Std: {submission['prediction'].std():.6f}\")\n        \n        # Create final submission with best strategy (weighted average)\n        if TOP_K_FEATURES == 80:\n            best_predictions = ensemble_strategies['weighted']\n            best_predictions = post_process_predictions(best_predictions)\n            \n            final_submission = pd.DataFrame({\n                'ID': test_ids,\n                'prediction': best_predictions\n            })\n            \n            final_submission.to_csv('submission.csv', index=False)\n            print(f\"\\n✅ Main submission saved to 'submission.csv'\")\n            \n            # Plot distribution\n            plt.figure(figsize=(10, 6))\n            plt.hist(final_submission['prediction'], bins=50, alpha=0.7, edgecolor='black')\n            plt.xlabel('Predicted Values')\n            plt.ylabel('Frequency')\n            plt.title('Distribution of Final Ensemble Predictions')\n            plt.grid(True, alpha=0.3)\n            plt.savefig('outputs/final_prediction_distribution.png', dpi=300, bbox_inches='tight')\n            plt.close()\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"ANALYSIS COMPLETED SUCCESSFULLY!\")\n    print(\"=\"*60)\n    print(\"\\nFiles created:\")\n    print(\"- submission.csv (main submission - 80 features, weighted ensemble)\")\n    print(\"- submission_80features_*.csv (different ensemble strategies)\")\n    print(\"- submission_100features_*.csv (different ensemble strategies)\")\n    print(\"- outputs/tangos_feature_importance_*.csv\")\n    print(\"- outputs/top_*_features.txt\")\n    print(\"- outputs/final_prediction_distribution.png\")\n    \n    return final_submission\n\n# =========================\n# Execute Analysis\n# =========================\nif __name__ == \"__main__\":\n    final_submission = main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}