{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DOFEN (Deep Oblivious Forest ENsemble) Feature Importance Analysis - FIXED VERSION\n# State-of-the-art tree-neural hybrid for tabular data\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import spearmanr, pearsonr\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nimport warnings\nimport os\nimport math\n\nwarnings.filterwarnings(\"ignore\")\n\n# Set style for better visualizations\nplt.style.use('seaborn-v0_8-darkgrid')\nsns.set_palette(\"husl\")\n\n# Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Feature Engineering Function\n# =========================\ndef add_features(df):\n    \"\"\"Add all engineered features with numerical stability\"\"\"\n    eps = 1e-8\n    \n    # Make a copy to avoid modifying original\n    df = df.copy()\n    \n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + eps)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + eps)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + eps)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + eps)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + eps)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + eps)\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + eps)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + eps)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + eps)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + eps)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + eps)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * np.log1p(df['volume'])\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + eps)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + eps)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + eps)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + eps)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + eps)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * np.log1p(df['total_depth'])\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * np.log1p(df['volume'])\n    df['depth_volume_interaction'] = np.log1p(df['total_depth']) * np.log1p(df['volume'])\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + eps)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + eps)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + eps) * np.log1p(df['volume'])\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + eps)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + eps)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + eps)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'] + eps)\n    df['sqrt_depth'] = np.sqrt(df['total_depth'] + eps)\n    df['volume_squared'] = np.minimum(df['volume'] ** 2, 1e10)\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + eps)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + eps)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + eps)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + eps)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + eps)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + eps) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + eps)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + eps)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * np.log1p(df['volume'])\n    \n    # Clip extreme values\n    numeric_cols = df.select_dtypes(include=[np.number]).columns\n    for col in numeric_cols:\n        if col not in ['label', 'timestamp']:\n            df[col] = np.clip(df[col], -1e6, 1e6)\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    return df\n\n# =========================\n# FIXED DOFEN Model Components\n# =========================\n\nclass StableObliviousDecisionTree(nn.Module):\n    \"\"\"Numerically stable Differentiable Oblivious Decision Tree\"\"\"\n    def __init__(self, depth, num_features, temperature=1.0):\n        super().__init__()\n        self.depth = depth\n        self.num_features = num_features\n        self.temperature = max(temperature, 0.1)  # Ensure minimum temperature\n        self.num_leaves = 2 ** depth\n        \n        # Better initialization for stability\n        self.feature_indices = nn.Parameter(torch.randn(depth, num_features) * 0.1)\n        self.thresholds = nn.Parameter(torch.zeros(depth))\n        self.leaf_values = nn.Parameter(torch.randn(self.num_leaves) * 0.1)\n        \n        # Add batch normalization for stability\n        self.feature_norm = nn.BatchNorm1d(num_features)\n        \n    def forward(self, x):\n        batch_size = x.shape[0]\n        \n        # Normalize features for stability\n        x_norm = self.feature_norm(x)\n        \n        # Compute leaf probabilities with numerical stability\n        leaf_log_probs = torch.zeros(batch_size, self.num_leaves, device=x.device)\n        \n        for d in range(self.depth):\n            # Stable feature selection using log-softmax\n            feature_log_weights = F.log_softmax(self.feature_indices[d] / self.temperature, dim=0)\n            feature_weights = torch.exp(feature_log_weights)\n            \n            # Select features with stability\n            selected_feature = (x_norm * feature_weights.unsqueeze(0)).sum(dim=1)\n            \n            # Stable sigmoid computation\n            decision_logit = (selected_feature - self.thresholds[d]) / self.temperature\n            decision_logit = torch.clamp(decision_logit, -10, 10)  # Prevent overflow\n            decision = torch.sigmoid(decision_logit)\n            \n            # Update leaf probabilities in log space for stability\n            for leaf in range(self.num_leaves):\n                if (leaf >> (self.depth - 1 - d)) & 1:\n                    leaf_log_probs[:, leaf] += torch.log(decision + 1e-8)\n                else:\n                    leaf_log_probs[:, leaf] += torch.log(1 - decision + 1e-8)\n        \n        # Convert back from log space\n        leaf_probs = torch.exp(leaf_log_probs)\n        leaf_probs = leaf_probs / (leaf_probs.sum(dim=1, keepdim=True) + 1e-8)\n        \n        # Compute output with bounded leaf values\n        bounded_leaf_values = torch.tanh(self.leaf_values)\n        output = (leaf_probs * bounded_leaf_values.unsqueeze(0)).sum(dim=1)\n        \n        return output\n    \n    def get_feature_importance(self):\n        \"\"\"Extract feature importance from the tree\"\"\"\n        importance = torch.zeros(self.num_features)\n        for d in range(self.depth):\n            feature_probs = F.softmax(self.feature_indices[d] / self.temperature, dim=0)\n            importance += feature_probs.detach().cpu()\n        return importance.numpy() / self.depth\n\nclass SimplifiedDOFEN(nn.Module):\n    \"\"\"Simplified and stable DOFEN model\"\"\"\n    def __init__(self, num_features, num_trees=20, tree_depth=3, temperature=2.0, dropout=0.3):\n        super().__init__()\n        self.num_features = num_features\n        self.num_trees = num_trees\n        self.tree_depth = tree_depth\n        \n        # Input normalization\n        self.input_norm = nn.BatchNorm1d(num_features)\n        self.input_dropout = nn.Dropout(dropout)\n        \n        # Forest of trees\n        self.trees = nn.ModuleList([\n            StableObliviousDecisionTree(tree_depth, num_features, temperature)\n            for _ in range(num_trees)\n        ])\n        \n        # Tree combination\n        self.tree_weights = nn.Parameter(torch.ones(num_trees) / num_trees)\n        \n        # Output head with residual connections\n        self.output_head = nn.Sequential(\n            nn.Linear(num_trees + 1, 32),  # +1 for residual\n            nn.BatchNorm1d(32),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(32, 16),\n            nn.BatchNorm1d(16),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(16, 1)\n        )\n        \n        # Direct linear path for residual\n        self.direct_linear = nn.Linear(num_features, 1)\n        \n        # Initialize weights properly\n        self._init_weights()\n        \n    def _init_weights(self):\n        \"\"\"Initialize weights for stability\"\"\"\n        for m in self.modules():\n            if isinstance(m, nn.Linear):\n                nn.init.xavier_uniform_(m.weight, gain=0.5)\n                if m.bias is not None:\n                    nn.init.constant_(m.bias, 0)\n    \n    def forward(self, x):\n        # Input normalization and dropout\n        x_norm = self.input_norm(x)\n        x_norm = self.input_dropout(x_norm)\n        \n        # Get tree outputs\n        tree_outputs = []\n        for tree in self.trees:\n            tree_out = tree(x_norm)\n            tree_outputs.append(tree_out)\n        \n        # Stack tree outputs\n        tree_outputs = torch.stack(tree_outputs, dim=1)  # [batch, num_trees]\n        \n        # Direct linear path (residual connection)\n        direct_out = self.direct_linear(x_norm)\n        \n        # Combine tree outputs with residual\n        combined = torch.cat([tree_outputs, direct_out], dim=1)  # [batch, num_trees + 1]\n        \n        # Final output\n        output = self.output_head(combined)\n        \n        return output.squeeze(-1)\n    \n    def get_tree_importance(self):\n        \"\"\"Get feature importance from all trees\"\"\"\n        importance = np.zeros(self.num_features)\n        tree_weights = F.softmax(self.tree_weights, dim=0).detach().cpu().numpy()\n        \n        for i, tree in enumerate(self.trees):\n            tree_importance = tree.get_feature_importance()\n            importance += tree_weights[i] * tree_importance\n        \n        return importance\n    \n    def get_gradient_importance(self, dataloader, criterion, device, max_batches=50):\n        \"\"\"Compute gradient-based feature importance\"\"\"\n        self.eval()\n        gradient_importance = torch.zeros(self.num_features)\n        n_batches = 0\n        \n        for inputs, targets in dataloader:\n            if n_batches >= max_batches:\n                break\n                \n            inputs = inputs.to(device).requires_grad_(True)\n            targets = targets.to(device)\n            \n            # Forward pass\n            outputs = self(inputs)\n            loss = criterion(outputs, targets.squeeze())\n            \n            # Backward pass\n            loss.backward()\n            \n            # Accumulate absolute gradients\n            if inputs.grad is not None:\n                gradient_importance += torch.abs(inputs.grad).mean(dim=0).cpu()\n            \n            n_batches += 1\n            \n            # Clear gradients\n            if inputs.grad is not None:\n                inputs.grad.zero_()\n        \n        if n_batches > 0:\n            gradient_importance /= n_batches\n            \n        return gradient_importance.numpy()\n\n# =========================\n# Training Function\n# =========================\n\ndef train_dofen_fixed(model, train_loader, val_loader, n_epochs=20, learning_rate=0.001, patience=5):\n    \"\"\"Train DOFEN model with better stability\"\"\"\n    criterion = nn.MSELoss()\n    optimizer = torch.optim.Adam(model.parameters(), lr=learning_rate, weight_decay=0.01)\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer, mode='min', factor=0.5, patience=3, min_lr=1e-6\n    )\n    \n    best_val_loss = float('inf')\n    patience_counter = 0\n    train_losses = []\n    val_losses = []\n    \n    for epoch in range(n_epochs):\n        # Training\n        model.train()\n        train_loss = 0.0\n        n_train_batches = 0\n        \n        progress_bar = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{n_epochs}\")\n        for inputs, targets in progress_bar:\n            inputs, targets = inputs.to(device), targets.to(device).squeeze()\n            \n            optimizer.zero_grad()\n            outputs = model(inputs)\n            loss = criterion(outputs, targets)\n            \n            # Skip if loss is NaN\n            if torch.isnan(loss):\n                continue\n            \n            loss.backward()\n            \n            # Gradient clipping for stability\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n            \n            optimizer.step()\n            \n            train_loss += loss.item()\n            n_train_batches += 1\n            progress_bar.set_postfix({'loss': f'{loss.item():.4f}'})\n        \n        # Validation\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        n_val_batches = 0\n        \n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs, targets = inputs.to(device), targets.to(device).squeeze()\n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n                \n                if not torch.isnan(loss):\n                    val_loss += loss.item()\n                    val_preds.extend(outputs.cpu().numpy())\n                    val_targets.extend(targets.cpu().numpy())\n                    n_val_batches += 1\n        \n        # Calculate metrics\n        avg_train_loss = train_loss / n_train_batches if n_train_batches > 0 else float('inf')\n        avg_val_loss = val_loss / n_val_batches if n_val_batches > 0 else float('inf')\n        \n        train_losses.append(avg_train_loss)\n        val_losses.append(avg_val_loss)\n        \n        if len(val_preds) > 0 and len(val_targets) > 0:\n            val_pearson = pearsonr(val_targets, val_preds)[0]\n            if np.isnan(val_pearson):\n                val_pearson = 0.0\n        else:\n            val_pearson = 0.0\n        \n        print(f\"Epoch {epoch+1}: Train Loss = {avg_train_loss:.4f}, \"\n              f\"Val Loss = {avg_val_loss:.4f}, Val Pearson = {val_pearson:.4f}\")\n        \n        scheduler.step(avg_val_loss)\n        \n        # Early stopping\n        if avg_val_loss < best_val_loss:\n            best_val_loss = avg_val_loss\n            patience_counter = 0\n            torch.save(model.state_dict(), 'best_dofen_model.pt')\n        else:\n            patience_counter += 1\n            if patience_counter >= patience:\n                print(\"Early stopping triggered\")\n                break\n    \n    # Load best model\n    if os.path.exists('best_dofen_model.pt'):\n        model.load_state_dict(torch.load('best_dofen_model.pt'))\n    \n    return model, train_losses, val_losses\n\n# =========================\n# Fixed Visualization Functions\n# =========================\n\ndef create_dofen_visualizations_fixed(importance_results, feature_names, save_prefix=\"dofen\"):\n    \"\"\"Create visualizations with proper NaN handling\"\"\"\n    \n    # Create output directory\n    os.makedirs('dofen_outputs', exist_ok=True)\n    \n    # 1. Combined importance horizontal bar plot\n    plt.figure(figsize=(14, 10))\n    top_n = 40\n    \n    combined_importance = importance_results['combined']\n    # Handle NaN values\n    combined_importance = np.nan_to_num(combined_importance, nan=0.0)\n    \n    top_indices = np.argsort(combined_importance)[-top_n:][::-1]\n    \n    top_features = [feature_names[i] for i in top_indices]\n    top_importance = combined_importance[top_indices]\n    \n    # Color coding\n    colors = []\n    for f in top_features:\n        if f.startswith('X'):\n            colors.append('#FF6B6B')  # Red for anonymous\n        elif f in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']:\n            colors.append('#4ECDC4')  # Teal for market\n        else:\n            colors.append('#45B7D1')  # Blue for engineered\n    \n    plt.barh(range(len(top_features)), top_importance, color=colors)\n    plt.yticks(range(len(top_features)), top_features)\n    plt.xlabel('Combined Importance Score', fontsize=12)\n    plt.title(f'Top {top_n} Most Important Features (DOFEN)', fontsize=16, fontweight='bold')\n    plt.gca().invert_yaxis()\n    \n    # Add legend\n    from matplotlib.patches import Patch\n    legend_elements = [\n        Patch(facecolor='#FF6B6B', label='Anonymous Features (X_)'),\n        Patch(facecolor='#4ECDC4', label='Market Features'),\n        Patch(facecolor='#45B7D1', label='Engineered Features')\n    ]\n    plt.legend(handles=legend_elements, loc='lower right')\n    \n    plt.tight_layout()\n    plt.savefig(f'dofen_outputs/{save_prefix}_top_features.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # 2. Component importance comparison\n    fig, axes = plt.subplots(1, 2, figsize=(14, 6))\n    \n    # Tree importance distribution\n    ax = axes[0]\n    tree_imp = np.nan_to_num(importance_results['tree_importance'], nan=0.0)\n    valid_tree_imp = tree_imp[tree_imp > 0]\n    \n    if len(valid_tree_imp) > 0:\n        ax.hist(valid_tree_imp, bins=30, alpha=0.7, color='forestgreen', edgecolor='black')\n        ax.axvline(valid_tree_imp.mean(), color='darkgreen', linestyle='--',\n                   label=f'Mean: {valid_tree_imp.mean():.4f}')\n        ax.set_xlabel('Tree-based Importance')\n        ax.set_ylabel('Frequency')\n        ax.set_title('Tree Structure Importance Distribution')\n        ax.legend()\n        ax.grid(True, alpha=0.3)\n    \n    # Gradient importance distribution\n    ax = axes[1]\n    grad_imp = np.nan_to_num(importance_results['gradient_importance'], nan=0.0)\n    valid_grad_imp = grad_imp[grad_imp > 0]\n    \n    if len(valid_grad_imp) > 0:\n        ax.hist(valid_grad_imp, bins=30, alpha=0.7, color='lightcoral', edgecolor='black')\n        ax.axvline(valid_grad_imp.mean(), color='darkred', linestyle='--',\n                   label=f'Mean: {valid_grad_imp.mean():.4f}')\n        ax.set_xlabel('Gradient-based Importance')\n        ax.set_ylabel('Frequency')\n        ax.set_title('Gradient Importance Distribution')\n        ax.legend()\n        ax.grid(True, alpha=0.3)\n    \n    plt.suptitle('DOFEN Feature Importance Components', fontsize=16, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig(f'dofen_outputs/{save_prefix}_components.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # 3. Feature category analysis\n    plt.figure(figsize=(10, 8))\n    \n    # Categorize features\n    anonymous_mask = [f.startswith('X') for f in feature_names]\n    market_mask = [f in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume'] for f in feature_names]\n    engineered_mask = [not (anonymous_mask[i] or market_mask[i]) for i in range(len(feature_names))]\n    \n    # Calculate average importance by category\n    anonymous_imp = combined_importance[anonymous_mask].mean() if any(anonymous_mask) else 0\n    market_imp = combined_importance[market_mask].mean() if any(market_mask) else 0\n    engineered_imp = combined_importance[engineered_mask].mean() if any(engineered_mask) else 0\n    \n    categories = ['Anonymous\\nFeatures', 'Market\\nFeatures', 'Engineered\\nFeatures']\n    avg_importances = [anonymous_imp, market_imp, engineered_imp]\n    colors = ['#FF6B6B', '#4ECDC4', '#45B7D1']\n    \n    bars = plt.bar(categories, avg_importances, color=colors, alpha=0.7, edgecolor='black')\n    plt.ylabel('Average Importance Score', fontsize=12)\n    plt.title('Average Feature Importance by Category', fontsize=14, fontweight='bold')\n    plt.grid(True, alpha=0.3, axis='y')\n    \n    # Add value labels on bars\n    for bar, val in zip(bars, avg_importances):\n        plt.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.001,\n                 f'{val:.4f}', ha='center', va='bottom')\n    \n    plt.tight_layout()\n    plt.savefig(f'dofen_outputs/{save_prefix}_category_analysis.png', dpi=300, bbox_inches='tight')\n    plt.show()\n\n# =========================\n# Main Analysis Function (Fixed)\n# =========================\n\ndef analyze_dofen_feature_importance_fixed():\n    \"\"\"Fixed DOFEN feature importance analysis\"\"\"\n    print(\"=== DOFEN Feature Importance Analysis (Fixed Version) ===\")\n    print(\"Deep Oblivious Forest ENsemble - Stabilized Implementation\\n\")\n    \n    # Data paths\n    data_paths = [\n        \"/kaggle/input/drw-crypto-market-prediction/train.parquet\",\n        \"../input/drw-crypto-market-prediction/train.parquet\",\n        \"./data/train.parquet\"\n    ]\n    \n    train_path = None\n    for path in data_paths:\n        if os.path.exists(path):\n            train_path = path\n            break\n    \n    if train_path is None:\n        train_path = data_paths[0]\n    \n    print(f\"Data path: {train_path}\")\n    \n    # Define features\n    x_features = [f\"X{i}\" for i in range(1, 891)]\n    market_features = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    \n    # Load data\n    try:\n        print(\"Loading data...\")\n        train_df = pd.read_parquet(train_path)\n        print(f\"Loaded {len(train_df)} samples\")\n        \n        # Check for NaN values in label\n        if train_df['label'].isna().sum() > 0:\n            print(f\"Found {train_df['label'].isna().sum()} NaN values in label, removing...\")\n            train_df = train_df[~train_df['label'].isna()]\n    except:\n        print(\"Could not load data. Generating synthetic data...\")\n        # Generate synthetic data\n        n_samples = 20000\n        np.random.seed(42)\n        \n        data = {}\n        # Create correlated features\n        base_features = np.random.randn(n_samples, 20)\n        \n        for i, feat in enumerate(x_features):\n            if i < 20:\n                # Important features\n                data[feat] = base_features[:, i % 20] + np.random.randn(n_samples) * 0.3\n            elif i < 50:\n                # Interaction features\n                idx1, idx2 = i % 20, (i + 5) % 20\n                data[feat] = base_features[:, idx1] * base_features[:, idx2] + np.random.randn(n_samples) * 0.5\n            else:\n                # Noise features\n                data[feat] = np.random.randn(n_samples) * 0.5\n        \n        # Market features\n        data['bid_qty'] = np.abs(np.random.lognormal(6, 1, n_samples))\n        data['ask_qty'] = np.abs(np.random.lognormal(6, 1, n_samples))\n        data['buy_qty'] = np.abs(np.random.lognormal(5, 1, n_samples))\n        data['sell_qty'] = np.abs(np.random.lognormal(5, 1, n_samples))\n        data['volume'] = data['buy_qty'] + data['sell_qty'] + np.abs(np.random.randn(n_samples) * 100)\n        \n        # Create label with tree-friendly patterns\n        data['label'] = (\n            0.3 * (data['X1'] > 0) * data['X1'] + \n            0.2 * (data['X2'] < 0) * data['X2'] + \n            0.15 * np.where(data['X3'] > np.median(data['X3']), data['X3'], -data['X3']) +\n            0.1 * np.log1p(data['volume']) +\n            0.05 * (data['buy_qty'] - data['sell_qty']) / (data['volume'] + 1) +\n            np.random.randn(n_samples) * 0.1\n        )\n        \n        train_df = pd.DataFrame(data)\n    \n    # Add engineered features\n    print(\"\\nAdding engineered features...\")\n    train_df = add_features(train_df)\n    \n    # Get all feature names\n    engineered_features = [col for col in train_df.columns \n                          if col not in x_features + market_features + ['label', 'timestamp']]\n    \n    all_features = x_features + market_features + engineered_features\n    print(f\"Total features: {len(all_features)}\")\n    \n    # Prepare data\n    print(\"\\nPreparing data...\")\n    sample_size = min(30000, int(0.5 * len(train_df)))\n    train_data = train_df.iloc[-sample_size:].reset_index(drop=True)\n    \n    X = train_data[all_features].values\n    y = train_data[\"label\"].values\n    \n    # Handle NaN values\n    X = np.nan_to_num(X, nan=0.0, posinf=1e6, neginf=-1e6)\n    y = np.nan_to_num(y, nan=0.0)\n    \n    # Split data\n    X_train, X_val, y_train, y_val = train_test_split(\n        X, y, test_size=0.2, random_state=42\n    )\n    \n    # Use StandardScaler for stability\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_val_scaled = scaler.transform(X_val)\n    \n    # Clip scaled values for additional stability\n    X_train_scaled = np.clip(X_train_scaled, -5, 5)\n    X_val_scaled = np.clip(X_val_scaled, -5, 5)\n    \n    # Create data loaders\n    train_dataset = TensorDataset(\n        torch.tensor(X_train_scaled, dtype=torch.float32),\n        torch.tensor(y_train, dtype=torch.float32)\n    )\n    val_dataset = TensorDataset(\n        torch.tensor(X_val_scaled, dtype=torch.float32),\n        torch.tensor(y_val, dtype=torch.float32)\n    )\n    \n    train_loader = DataLoader(train_dataset, batch_size=256, shuffle=True)\n    val_loader = DataLoader(val_dataset, batch_size=512, shuffle=False)\n    \n    # Initialize simplified DOFEN model\n    print(\"\\nInitializing simplified DOFEN model...\")\n    model = SimplifiedDOFEN(\n        num_features=len(all_features),\n        num_trees=15,\n        tree_depth=3,\n        temperature=2.0,\n        dropout=0.3\n    ).to(device)\n    \n    print(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n    print(f\"Architecture: 15 trees, depth 3, with residual connections\")\n    \n    # Train model\n    print(\"\\nTraining DOFEN model...\")\n    model, train_losses, val_losses = train_dofen_fixed(\n        model, train_loader, val_loader, n_epochs=15, learning_rate=0.001\n    )\n    \n    # Compute feature importance\n    print(\"\\n=== Computing DOFEN Feature Importance ===\")\n    \n    print(\"\\n1. Computing tree-based importance...\")\n    tree_importance = model.get_tree_importance()\n    \n    print(\"\\n2. Computing gradient-based importance...\")\n    criterion = nn.MSELoss()\n    gradient_importance = model.get_gradient_importance(val_loader, criterion, device)\n    \n    # Combine importance scores\n    print(\"\\n3. Combining importance scores...\")\n    \n    # Normalize importances\n    tree_importance = np.nan_to_num(tree_importance, nan=0.0)\n    gradient_importance = np.nan_to_num(gradient_importance, nan=0.0)\n    \n    if tree_importance.max() > tree_importance.min():\n        tree_importance = (tree_importance - tree_importance.min()) / (tree_importance.max() - tree_importance.min())\n    \n    if gradient_importance.max() > gradient_importance.min():\n        gradient_importance = (gradient_importance - gradient_importance.min()) / (gradient_importance.max() - gradient_importance.min())\n    \n    # Combined importance (weighted average)\n    combined_importance = 0.6 * tree_importance + 0.4 * gradient_importance\n    \n    importance_dict = {\n        'tree_importance': tree_importance,\n        'gradient_importance': gradient_importance,\n        'combined': combined_importance\n    }\n    \n    # Create visualizations\n    print(\"\\n=== Creating Visualizations ===\")\n    create_dofen_visualizations_fixed(importance_dict, all_features)\n    \n    # Create results dataframe\n    results_df = pd.DataFrame({\n        'feature': all_features,\n        'combined_importance': combined_importance,\n        'tree_importance': tree_importance,\n        'gradient_importance': gradient_importance\n    })\n    \n    # Sort by combined importance\n    results_df = results_df.sort_values('combined_importance', ascending=False)\n    \n    # Display results\n    print(\"\\n\" + \"=\"*80)\n    print(\"DOFEN FEATURE IMPORTANCE RESULTS\")\n    print(\"=\"*80)\n    \n    print(\"\\n🔍 Top 30 Most Important Features:\")\n    print(\"-\" * 60)\n    for idx, row in results_df.head(30).iterrows():\n        print(f\"{row['feature']:30s} {row['combined_importance']:.6f}\")\n    \n    # Category analysis\n    print(\"\\n📊 Feature Importance by Category:\")\n    print(\"-\" * 60)\n    \n    # Market features\n    market_df = results_df[results_df['feature'].isin(market_features)]\n    print(f\"\\nMarket Features (n={len(market_df)}):\")\n    for idx, row in market_df.iterrows():\n        print(f\"  {row['feature']:25s} Combined: {row['combined_importance']:.4f}\")\n    \n    # Top engineered features\n    engineered_df = results_df[results_df['feature'].isin(engineered_features)]\n    print(f\"\\nTop 15 Engineered Features (n={len(engineered_df)}):\")\n    for idx, row in engineered_df.head(15).iterrows():\n        print(f\"  {row['feature']:25s} Combined: {row['combined_importance']:.4f}\")\n    \n    # Top anonymous features\n    x_df = results_df[results_df['feature'].str.startswith('X')]\n    print(f\"\\nTop 15 Anonymous Features (n={len(x_df)}):\")\n    for idx, row in x_df.head(15).iterrows():\n        print(f\"  {row['feature']:25s} Combined: {row['combined_importance']:.4f}\")\n    \n    # Save results\n    os.makedirs('dofen_outputs', exist_ok=True)\n    results_df.to_csv(\"dofen_outputs/dofen_feature_importance_fixed.csv\", index=False)\n    print(\"\\n✅ Results saved to 'dofen_outputs/dofen_feature_importance_fixed.csv'\")\n    \n    # Training curve plot\n    if len(train_losses) > 0 and len(val_losses) > 0:\n        plt.figure(figsize=(10, 6))\n        epochs = range(1, len(train_losses) + 1)\n        plt.plot(epochs, train_losses, 'b-', label='Training Loss')\n        plt.plot(epochs, val_losses, 'r-', label='Validation Loss')\n        plt.xlabel('Epoch')\n        plt.ylabel('Loss')\n        plt.title('DOFEN Training Progress', fontsize=14, fontweight='bold')\n        plt.legend()\n        plt.grid(True, alpha=0.3)\n        plt.tight_layout()\n        plt.savefig('dofen_outputs/dofen_training_progress.png', dpi=300)\n        plt.show()\n    \n    return results_df, importance_dict\n\n# =========================\n# Execute Analysis\n# =========================\n\nif __name__ == \"__main__\":\n    print(\"Starting Fixed DOFEN Feature Importance Analysis...\")\n    print(\"=\"*80)\n    \n    # Set random seeds for reproducibility\n    np.random.seed(42)\n    torch.manual_seed(42)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(42)\n    \n    results_df, importance_dict = analyze_dofen_feature_importance_fixed()\n    \n    print(\"\\n✅ DOFEN Analysis completed successfully!\")\n    print(\"Check the 'dofen_outputs' directory for results and visualizations.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}