{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enhanced DCN V2 Feature Importance Analysis with Visualizations\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import spearmanr, pearsonr\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nimport itertools\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Set style for better visualizations\nplt.style.use('seaborn-v0_8-darkgrid')\nsns.set_palette(\"husl\")\n\n# Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Feature Engineering Function\n# =========================\ndef add_features(df):\n    \"\"\"Add all engineered features\"\"\"\n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df\n\n# =========================\n# DCN V2 Model Components\n# =========================\nclass CrossLayer(nn.Module):\n    \"\"\"Single cross layer in DCN V2 with matrix parameterization\"\"\"\n    def __init__(self, input_dim, use_bias=True):\n        super().__init__()\n        self.input_dim = input_dim\n        \n        # Weight matrix for cross operation\n        self.weight = nn.Parameter(torch.empty(input_dim, input_dim))\n        nn.init.xavier_normal_(self.weight)\n        \n        # Bias\n        self.use_bias = use_bias\n        if use_bias:\n            self.bias = nn.Parameter(torch.zeros(input_dim))\n            \n    def forward(self, x0, x):\n        # x0: initial input, x: current layer input\n        # Cross operation: x0 * (W * x^T) + bias + x\n        batch_size = x.shape[0]\n        \n        # Compute x^T * W\n        feature_cross = torch.matmul(x, self.weight)  # (batch_size, input_dim)\n        \n        # Element-wise multiplication with x0\n        cross_output = x0 * feature_cross\n        \n        if self.use_bias:\n            cross_output = cross_output + self.bias\n            \n        # Residual connection\n        output = cross_output + x\n        \n        return output\n    \n    def get_cross_weights(self):\n        \"\"\"Get the cross weight matrix for analysis\"\"\"\n        return self.weight.detach().cpu().numpy()\n\nclass MixtureCrossLayer(nn.Module):\n    \"\"\"Mixture of Experts Cross Layer for efficiency\"\"\"\n    def __init__(self, input_dim, num_experts=4, low_rank=32, use_bias=True):\n        super().__init__()\n        self.input_dim = input_dim\n        self.num_experts = num_experts\n        self.low_rank = low_rank\n        \n        # Expert weights (low-rank factorization)\n        self.U = nn.Parameter(torch.empty(input_dim, low_rank, num_experts))\n        self.V = nn.Parameter(torch.empty(input_dim, low_rank, num_experts))\n        self.C = nn.Parameter(torch.empty(input_dim, low_rank, num_experts))\n        \n        nn.init.xavier_normal_(self.U)\n        nn.init.xavier_normal_(self.V)\n        nn.init.xavier_normal_(self.C)\n        \n        # Gating network\n        self.gate = nn.Sequential(\n            nn.Linear(input_dim, num_experts),\n            nn.Softmax(dim=1)\n        )\n        \n        self.use_bias = use_bias\n        if use_bias:\n            self.bias = nn.Parameter(torch.zeros(input_dim))\n            \n    def forward(self, x0, x):\n        batch_size = x.shape[0]\n        \n        # Compute expert gates\n        gates = self.gate(x)  # (batch_size, num_experts)\n        \n        # Compute expert outputs\n        expert_outputs = []\n        for i in range(self.num_experts):\n            # Low-rank cross: x0 * (U * V^T * x) where U, V are expert-specific\n            Ui = self.U[:, :, i]  # (input_dim, low_rank)\n            Vi = self.V[:, :, i]  # (input_dim, low_rank)\n            Ci = self.C[:, :, i]  # (input_dim, low_rank)\n            \n            # Efficient computation\n            v_x = torch.matmul(x, Vi)  # (batch_size, low_rank)\n            u_v_x = torch.matmul(v_x, Ui.t())  # (batch_size, input_dim)\n            \n            # Apply gating implicitly\n            c_x = torch.matmul(x, Ci)  # (batch_size, low_rank)\n            \n            # Cross output\n            cross_i = x0 * u_v_x * gates[:, i:i+1]\n            expert_outputs.append(cross_i)\n        \n        # Combine expert outputs\n        cross_output = sum(expert_outputs)\n        \n        if self.use_bias:\n            cross_output = cross_output + self.bias\n            \n        # Residual connection\n        output = cross_output + x\n        \n        return output\n    \n    def get_expert_importance(self, x):\n        \"\"\"Get importance of each expert\"\"\"\n        gates = self.gate(x)\n        return gates.mean(dim=0).detach().cpu().numpy()\n\nclass CrossNetwork(nn.Module):\n    \"\"\"Cross Network with multiple cross layers\"\"\"\n    def __init__(self, input_dim, num_layers=3, use_mixture=True, num_experts=4):\n        super().__init__()\n        self.input_dim = input_dim\n        self.num_layers = num_layers\n        self.use_mixture = use_mixture\n        \n        self.cross_layers = nn.ModuleList()\n        for i in range(num_layers):\n            if use_mixture:\n                layer = MixtureCrossLayer(input_dim, num_experts=num_experts)\n            else:\n                layer = CrossLayer(input_dim)\n            self.cross_layers.append(layer)\n            \n    def forward(self, x):\n        x0 = x\n        for layer in self.cross_layers:\n            x = layer(x0, x)\n        return x\n    \n    def get_cross_weights(self):\n        \"\"\"Get cross weights from all layers\"\"\"\n        weights = []\n        for i, layer in enumerate(self.cross_layers):\n            if isinstance(layer, CrossLayer):\n                weights.append(layer.get_cross_weights())\n        return weights\n    \n    def get_interaction_strengths(self):\n        \"\"\"Compute feature interaction strengths\"\"\"\n        if not self.use_mixture:\n            # For matrix cross layers\n            interaction_matrix = np.zeros((self.input_dim, self.input_dim))\n            \n            for layer in self.cross_layers:\n                if isinstance(layer, CrossLayer):\n                    W = layer.get_cross_weights()\n                    interaction_matrix += np.abs(W)\n                    \n            return interaction_matrix / self.num_layers\n        else:\n            # For mixture layers, return None (handled differently)\n            return None\n\nclass DeepNetwork(nn.Module):\n    \"\"\"Deep network component of DCN V2\"\"\"\n    def __init__(self, input_dim, hidden_dims=[256, 128, 64], dropout=0.2):\n        super().__init__()\n        \n        layers = []\n        prev_dim = input_dim\n        \n        for hidden_dim in hidden_dims:\n            layers.extend([\n                nn.Linear(prev_dim, hidden_dim),\n                nn.BatchNorm1d(hidden_dim),\n                nn.ReLU(),\n                nn.Dropout(dropout)\n            ])\n            prev_dim = hidden_dim\n            \n        self.network = nn.Sequential(*layers)\n        self.output_dim = prev_dim\n        \n    def forward(self, x):\n        return self.network(x)\n\nclass DCNV2(nn.Module):\n    \"\"\"DCN V2: Deep & Cross Network V2\"\"\"\n    def __init__(self, input_dim, num_cross_layers=3, deep_hidden_dims=[256, 128, 64],\n                 use_mixture=True, num_experts=4, combination='parallel'):\n        super().__init__()\n        \n        self.input_dim = input_dim\n        self.combination = combination  # 'parallel' or 'stacked'\n        \n        # Cross Network\n        self.cross_network = CrossNetwork(\n            input_dim, \n            num_layers=num_cross_layers,\n            use_mixture=use_mixture,\n            num_experts=num_experts\n        )\n        \n        # Deep Network\n        self.deep_network = DeepNetwork(input_dim, deep_hidden_dims)\n        \n        # Final layers\n        if combination == 'parallel':\n            # Parallel: concat cross and deep outputs\n            final_dim = input_dim + self.deep_network.output_dim\n        else:\n            # Stacked: cross output feeds into deep\n            final_dim = self.deep_network.output_dim\n            \n        self.final_layer = nn.Sequential(\n            nn.Linear(final_dim, 64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(64, 1)\n        )\n        \n        # Feature importance tracking\n        self.register_buffer('feature_gradients', torch.zeros(input_dim))\n        self.register_buffer('gradient_count', torch.tensor(0))\n        \n    def forward(self, x, track_gradients=False):\n        if self.combination == 'parallel':\n            # Parallel structure\n            cross_output = self.cross_network(x)\n            deep_output = self.deep_network(x)\n            \n            # Concatenate\n            combined = torch.cat([cross_output, deep_output], dim=1)\n        else:\n            # Stacked structure\n            cross_output = self.cross_network(x)\n            deep_output = self.deep_network(cross_output)\n            combined = deep_output\n            \n        # Final prediction\n        output = self.final_layer(combined)\n        \n        if track_gradients and x.requires_grad:\n            self.accumulate_gradients(x, output)\n            \n        return output\n    \n    def accumulate_gradients(self, x, output):\n        \"\"\"Accumulate gradients for feature importance\"\"\"\n        grad_outputs = torch.ones_like(output)\n        grads = torch.autograd.grad(outputs=output, inputs=x, \n                                   grad_outputs=grad_outputs, \n                                   retain_graph=True, create_graph=True)[0]\n        \n        self.feature_gradients += torch.abs(grads).mean(dim=0).detach()\n        self.gradient_count += 1\n    \n    def get_gradient_importance(self):\n        \"\"\"Get accumulated gradient importance\"\"\"\n        if self.gradient_count > 0:\n            return (self.feature_gradients / self.gradient_count).cpu().numpy()\n        else:\n            return np.zeros(self.input_dim)\n    \n    def reset_gradient_importance(self):\n        \"\"\"Reset gradient accumulator\"\"\"\n        self.feature_gradients.zero_()\n        self.gradient_count.zero_()\n    \n    def compute_cross_importance(self):\n        \"\"\"Compute importance from cross network weights\"\"\"\n        if not self.cross_network.use_mixture:\n            # Get interaction matrix\n            interaction_matrix = self.cross_network.get_interaction_strengths()\n            \n            # Feature importance = sum of interactions with other features\n            cross_importance = interaction_matrix.sum(axis=0) + interaction_matrix.sum(axis=1)\n            return cross_importance / 2\n        else:\n            # For mixture layers, use different approach\n            return None\n    \n    def compute_feature_interactions(self, X, n_samples=3000, top_k=20):\n        \"\"\"Compute pairwise feature interactions through cross network\"\"\"\n        self.eval()\n        \n        # Use subset of data\n        if len(X) > n_samples:\n            indices = np.random.choice(len(X), n_samples, replace=False)\n            X_subset = X[indices]\n        else:\n            X_subset = X\n            \n        X_tensor = torch.tensor(X_subset, dtype=torch.float32, device=device)\n        \n        # Get baseline predictions\n        with torch.no_grad():\n            baseline_preds = self.forward(X_tensor).cpu().numpy().flatten()\n        \n        # Compute pairwise interactions for top features\n        # First, get individual feature importance\n        individual_importance = np.zeros(self.input_dim)\n        \n        for i in range(self.input_dim):\n            X_zeroed = X_subset.copy()\n            X_zeroed[:, i] = 0\n            \n            with torch.no_grad():\n                X_zeroed_tensor = torch.tensor(X_zeroed, dtype=torch.float32, device=device)\n                zeroed_preds = self.forward(X_zeroed_tensor).cpu().numpy().flatten()\n                \n            individual_importance[i] = np.abs(baseline_preds - zeroed_preds).mean()\n        \n        # Get top k features\n        top_features = np.argsort(individual_importance)[-top_k:][::-1]\n        \n        # Compute pairwise interactions for top features\n        interaction_matrix = np.zeros((top_k, top_k))\n        \n        for idx_i, i in enumerate(tqdm(top_features, desc=\"Computing interactions\")):\n            for idx_j, j in enumerate(top_features):\n                if i >= j:\n                    continue\n                    \n                # Zero out both features\n                X_both_zeroed = X_subset.copy()\n                X_both_zeroed[:, i] = 0\n                X_both_zeroed[:, j] = 0\n                \n                with torch.no_grad():\n                    X_both_tensor = torch.tensor(X_both_zeroed, dtype=torch.float32, device=device)\n                    both_zeroed_preds = self.forward(X_both_tensor).cpu().numpy().flatten()\n                \n                # Interaction = effect of zeroing both - sum of individual effects\n                interaction_effect = np.abs(baseline_preds - both_zeroed_preds).mean()\n                expected_effect = individual_importance[i] + individual_importance[j]\n                interaction_strength = abs(interaction_effect - expected_effect)\n                \n                interaction_matrix[idx_i, idx_j] = interaction_strength\n                interaction_matrix[idx_j, idx_i] = interaction_strength\n                \n        return top_features, interaction_matrix, individual_importance\n    \n    def compute_polynomial_importance(self, X, max_order=3, n_samples=2000):\n        \"\"\"Compute importance of polynomial feature combinations\"\"\"\n        self.eval()\n        \n        # Use subset of data\n        if len(X) > n_samples:\n            indices = np.random.choice(len(X), n_samples, replace=False)\n            X_subset = X[indices]\n        else:\n            X_subset = X\n            \n        polynomial_importance = defaultdict(float)\n        \n        # Get top features first\n        individual_importance = self.compute_feature_interactions(X_subset, n_samples=1000, top_k=10)[2]\n        top_features = np.argsort(individual_importance)[-10:][::-1]\n        \n        # Test polynomial combinations up to max_order\n        for order in range(1, min(max_order + 1, len(top_features) + 1)):\n            for feature_combo in itertools.combinations(top_features, order):\n                X_poly = X_subset.copy()\n                \n                # Create polynomial feature\n                poly_feature = np.ones(len(X_subset))\n                for feat_idx in feature_combo:\n                    poly_feature *= X_subset[:, feat_idx]\n                \n                # Standardize\n                poly_feature = (poly_feature - poly_feature.mean()) / (poly_feature.std() + 1e-8)\n                \n                # Test importance by correlation with predictions\n                X_tensor = torch.tensor(X_subset, dtype=torch.float32, device=device)\n                with torch.no_grad():\n                    preds = self.forward(X_tensor).cpu().numpy().flatten()\n                \n                correlation = abs(pearsonr(poly_feature, preds)[0])\n                polynomial_importance[feature_combo] = correlation\n                \n        return polynomial_importance\n\n# =========================\n# Visualization Functions\n# =========================\ndef create_dcnv2_visualizations(importance_results, feature_names, save_prefix=\"dcnv2\"):\n    \"\"\"Create comprehensive visualizations for DCN V2 feature importance\"\"\"\n    \n    # 1. Combined importance bar plot\n    plt.figure(figsize=(12, 8))\n    top_n = 30\n    \n    # Get combined importance\n    combined_importance = importance_results['combined_importance']\n    top_indices = np.argsort(combined_importance)[-top_n:][::-1]\n    \n    top_features = [feature_names[i] for i in top_indices]\n    top_importance = combined_importance[top_indices]\n    \n    # Color code by feature type\n    colors = ['#FF6B6B' if f.startswith('X') else '#4ECDC4' if f in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume'] else '#45B7D1' for f in top_features]\n    \n    bars = plt.bar(range(len(top_features)), top_importance, color=colors)\n    plt.xticks(range(len(top_features)), top_features, rotation=45, ha='right')\n    plt.xlabel('Features')\n    plt.ylabel('Combined Importance Score')\n    plt.title(f'Top {top_n} Most Important Features (DCN V2)', fontsize=16, fontweight='bold')\n    plt.tight_layout()\n    \n    # Add legend\n    from matplotlib.patches import Patch\n    legend_elements = [\n        Patch(facecolor='#FF6B6B', label='Anonymous Features (X_)'),\n        Patch(facecolor='#4ECDC4', label='Market Features'),\n        Patch(facecolor='#45B7D1', label='Engineered Features')\n    ]\n    plt.legend(handles=legend_elements, loc='upper right')\n    \n    plt.savefig(f'{save_prefix}_top_features.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # 2. Feature interaction heatmap\n    if 'interaction_matrix' in importance_results and importance_results['interaction_matrix'] is not None:\n        plt.figure(figsize=(12, 10))\n        \n        top_features = importance_results['interaction_top_features']\n        interaction_matrix = importance_results['interaction_matrix']\n        \n        # Create labels\n        labels = [feature_names[idx] for idx in top_features]\n        \n        # Plot heatmap\n        ax = sns.heatmap(interaction_matrix, \n                        xticklabels=labels,\n                        yticklabels=labels,\n                        cmap='YlOrRd',\n                        cbar_kws={'label': 'Interaction Strength'},\n                        square=True)\n        \n        plt.xticks(rotation=45, ha='right')\n        plt.yticks(rotation=0)\n        plt.title('Feature Interaction Matrix (Top Features)', fontsize=16, fontweight='bold')\n        plt.tight_layout()\n        plt.savefig(f'{save_prefix}_interaction_matrix.png', dpi=300, bbox_inches='tight')\n        plt.show()\n    \n    # 3. DCN V2 components visualization\n    fig, axes = plt.subplots(2, 2, figsize=(14, 10))\n    \n    # Cross importance\n    ax = axes[0, 0]\n    if 'cross_importance' in importance_results and importance_results['cross_importance'] is not None:\n        cross_importance = importance_results['cross_importance']\n        ax.hist(cross_importance, bins=50, alpha=0.7, color='skyblue', edgecolor='black')\n        ax.axvline(cross_importance.mean(), color='red', linestyle='--', \n                   label=f'Mean: {cross_importance.mean():.4f}')\n        ax.set_xlabel('Cross Network Importance')\n        ax.set_ylabel('Frequency')\n        ax.set_title('Cross Network Importance Distribution')\n        ax.legend()\n        ax.grid(True, alpha=0.3)\n    \n    # Gradient importance\n    ax = axes[0, 1]\n    gradient_importance = importance_results['gradient_importance']\n    ax.hist(gradient_importance, bins=50, alpha=0.7, color='lightcoral', edgecolor='black')\n    ax.axvline(gradient_importance.mean(), color='darkred', linestyle='--',\n               label=f'Mean: {gradient_importance.mean():.4f}')\n    ax.set_xlabel('Gradient Importance')\n    ax.set_ylabel('Frequency')\n    ax.set_title('Gradient Importance Distribution')\n    ax.legend()\n    ax.grid(True, alpha=0.3)\n    \n    # Individual feature effects\n    ax = axes[1, 0]\n    individual_effects = importance_results['individual_effects']\n    ax.hist(individual_effects, bins=50, alpha=0.7, color='lightgreen', edgecolor='black')\n    ax.axvline(individual_effects.mean(), color='darkgreen', linestyle='--',\n               label=f'Mean: {individual_effects.mean():.4f}')\n    ax.set_xlabel('Individual Feature Effect')\n    ax.set_ylabel('Frequency')\n    ax.set_title('Individual Feature Effect Distribution')\n    ax.legend()\n    ax.grid(True, alpha=0.3)\n    \n    # Top polynomial features\n    ax = axes[1, 1]\n    if 'polynomial_importance' in importance_results:\n        poly_importance = importance_results['polynomial_importance']\n        \n        # Get top polynomial features\n        top_polys = sorted(poly_importance.items(), key=lambda x: x[1], reverse=True)[:10]\n        \n        poly_names = []\n        poly_values = []\n        for combo, value in top_polys:\n            if len(combo) == 1:\n                name = feature_names[combo[0]]\n            else:\n                name = ' × '.join([feature_names[idx] for idx in combo])\n            poly_names.append(name[:30] + '...' if len(name) > 30 else name)\n            poly_values.append(value)\n        \n        y_pos = np.arange(len(poly_names))\n        ax.barh(y_pos, poly_values, color='plum')\n        ax.set_yticks(y_pos)\n        ax.set_yticklabels(poly_names)\n        ax.set_xlabel('Correlation with Output')\n        ax.set_title('Top Polynomial Features')\n        ax.grid(True, alpha=0.3, axis='x')\n    \n    plt.suptitle('DCN V2 Component Analysis', fontsize=16, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig(f'{save_prefix}_components.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # 4. Cross layer analysis\n    if not importance_results.get('use_mixture', True):\n        # For matrix cross layers, show weight evolution\n        plt.figure(figsize=(10, 6))\n        \n        # This would show how cross weights evolve through layers\n        # (Implementation depends on specific cross weight structure)\n        plt.text(0.5, 0.5, 'Cross Layer Weight Analysis\\n(Matrix Parameterization)', \n                ha='center', va='center', fontsize=16)\n        plt.axis('off')\n        plt.tight_layout()\n        plt.savefig(f'{save_prefix}_cross_analysis.png', dpi=300, bbox_inches='tight')\n        plt.show()\n    \n    # 5. Feature importance by category with interaction strength\n    plt.figure(figsize=(12, 6))\n    \n    # Categorize features\n    x_features_idx = [i for i, f in enumerate(feature_names) if f.startswith('X')]\n    market_features_idx = [i for i, f in enumerate(feature_names) if f in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']]\n    engineered_features_idx = [i for i in range(len(feature_names)) if i not in x_features_idx and i not in market_features_idx]\n    \n    categories = ['Anonymous (X_)', 'Market', 'Engineered']\n    \n    # Calculate average importance and interaction strength by category\n    avg_importance = [\n        combined_importance[x_features_idx].mean(),\n        combined_importance[market_features_idx].mean(),\n        combined_importance[engineered_features_idx].mean()\n    ]\n    \n    # Create bar plot\n    x = np.arange(len(categories))\n    width = 0.35\n    \n    fig, ax = plt.subplots(figsize=(10, 6))\n    bars1 = ax.bar(x - width/2, avg_importance, width, label='Average Importance', \n                    color=['#FF6B6B', '#4ECDC4', '#45B7D1'])\n    \n    # Add value labels\n    for bar in bars1:\n        height = bar.get_height()\n        ax.text(bar.get_x() + bar.get_width()/2., height,\n                f'{height:.4f}', ha='center', va='bottom')\n    \n    ax.set_xlabel('Feature Category')\n    ax.set_ylabel('Average Score')\n    ax.set_title('Feature Importance by Category', fontsize=16, fontweight='bold')\n    ax.set_xticks(x)\n    ax.set_xticklabels(categories)\n    ax.legend()\n    ax.grid(True, alpha=0.3, axis='y')\n    \n    plt.tight_layout()\n    plt.savefig(f'{save_prefix}_category_analysis.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # 6. Network architecture visualization\n    plt.figure(figsize=(10, 8))\n    \n    # Simple network architecture diagram\n    ax = plt.gca()\n    ax.text(0.5, 0.9, 'DCN V2 Architecture', fontsize=18, ha='center', fontweight='bold')\n    \n    # Input\n    ax.text(0.2, 0.7, 'Input Features', fontsize=14, ha='center', \n            bbox=dict(boxstyle=\"round,pad=0.3\", facecolor=\"lightblue\"))\n    \n    # Cross Network\n    ax.text(0.2, 0.5, 'Cross Network\\n(Explicit Interactions)', fontsize=12, ha='center',\n            bbox=dict(boxstyle=\"round,pad=0.3\", facecolor=\"lightgreen\"))\n    \n    # Deep Network\n    ax.text(0.5, 0.5, 'Deep Network\\n(Implicit Patterns)', fontsize=12, ha='center',\n            bbox=dict(boxstyle=\"round,pad=0.3\", facecolor=\"lightcoral\"))\n    \n    # Output\n    ax.text(0.35, 0.3, 'Combined Features', fontsize=12, ha='center')\n    ax.text(0.35, 0.1, 'Output', fontsize=14, ha='center',\n            bbox=dict(boxstyle=\"round,pad=0.3\", facecolor=\"lightyellow\"))\n    \n    # Arrows\n    ax.annotate('', xy=(0.2, 0.65), xytext=(0.2, 0.55),\n                arrowprops=dict(arrowstyle='->', lw=2, color='black'))\n    ax.annotate('', xy=(0.5, 0.65), xytext=(0.5, 0.55),\n                arrowprops=dict(arrowstyle='->', lw=2, color='black'))\n    ax.annotate('', xy=(0.3, 0.35), xytext=(0.2, 0.45),\n                arrowprops=dict(arrowstyle='->', lw=2, color='black'))\n    ax.annotate('', xy=(0.4, 0.35), xytext=(0.5, 0.45),\n                arrowprops=dict(arrowstyle='->', lw=2, color='black'))\n    ax.annotate('', xy=(0.35, 0.25), xytext=(0.35, 0.15),\n                arrowprops=dict(arrowstyle='->', lw=2, color='black'))\n    \n    ax.set_xlim(0, 0.7)\n    ax.set_ylim(0, 1)\n    ax.axis('off')\n    \n    plt.tight_layout()\n    plt.savefig(f'{save_prefix}_architecture.png', dpi=300, bbox_inches='tight')\n    plt.show()\n\n# =========================\n# Main Feature Importance Analysis\n# =========================\ndef analyze_dcnv2_feature_importance():\n    print(\"=== Enhanced DCN V2 Feature Importance Analysis ===\\n\")\n    \n    # Load data\n    print(\"Loading data...\")\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    \n    # Get all X features\n    x_features = [f\"X{i}\" for i in range(1, 891)]\n    market_features = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    \n    # Load data with all features\n    train_df = pd.read_parquet(train_path, columns=x_features + market_features + [\"label\"])\n    print(f\"Loaded {len(train_df)} samples\")\n    \n    # Add engineered features\n    print(\"\\nAdding engineered features...\")\n    train_df = add_features(train_df)\n    \n    # Get all feature names\n    all_features = x_features + market_features\n    \n    # Engineered features\n    engineered_features = [\n        \"log_volume\", 'bid_ask_interaction', 'bid_buy_interaction', 'bid_sell_interaction', \n        'ask_buy_interaction', 'ask_sell_interaction', 'net_order_flow', 'normalized_net_flow',\n        'buying_pressure', 'volume_weighted_buy', 'total_depth', 'depth_imbalance',\n        'relative_spread', 'log_depth', 'kyle_lambda', 'flow_toxicity', 'aggressive_flow_ratio',\n        'volume_depth_ratio', 'activity_intensity', 'log_buy_qty', 'log_sell_qty',\n        'log_bid_qty', 'log_ask_qty', 'realized_spread_proxy', 'price_impact_proxy',\n        'quote_volatility_proxy', 'flow_depth_interaction', 'imbalance_volume_interaction',\n        'depth_volume_interaction', 'buy_sell_spread', 'bid_ask_spread', 'trade_informativeness',\n        'execution_shortfall_proxy', 'adverse_selection_proxy', 'fill_probability',\n        'execution_rate', 'market_efficiency', 'sqrt_volume', 'sqrt_depth', 'volume_squared',\n        'imbalance_squared', 'bid_ratio', 'ask_ratio', 'buy_ratio', 'sell_ratio',\n        'liquidity_consumption', 'market_stress', 'depth_depletion', 'net_buying_ratio',\n        'directional_volume', 'signed_volume', 'volume_weighted_sell', 'buy_sell_ratio',\n        'selling_pressure', 'effective_spread_proxy', 'bid_ask_imbalance', 'order_flow_imbalance',\n        'liquidity_ratio'\n    ]\n    \n    all_features = all_features + engineered_features\n    print(f\"Total features: {len(all_features)}\")\n    \n    # Prepare data - Use more data for robustness\n    print(\"\\nPreparing data...\")\n    train_size = int(0.5 * len(train_df))  # Use 50% of data\n    train_data = train_df.iloc[-train_size:].reset_index(drop=True)\n    \n    X = train_data[all_features].values\n    y = train_data[\"label\"].values\n    \n    # Split data\n    X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n    \n    # Scale data\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_val_scaled = scaler.transform(X_val)\n    \n    # Create data loaders\n    train_dataset = TensorDataset(\n        torch.tensor(X_train_scaled, dtype=torch.float32),\n        torch.tensor(y_train, dtype=torch.float32).unsqueeze(1)\n    )\n    val_dataset = TensorDataset(\n        torch.tensor(X_val_scaled, dtype=torch.float32),\n        torch.tensor(y_val, dtype=torch.float32).unsqueeze(1)\n    )\n    \n    train_loader = DataLoader(train_dataset, batch_size=512, shuffle=True)\n    val_loader = DataLoader(val_dataset, batch_size=1024, shuffle=False)\n    \n    # Initialize DCN V2 model\n    print(\"\\nInitializing DCN V2 model...\")\n    use_mixture = True  # Use mixture of experts for efficiency\n    \n    model = DCNV2(\n        input_dim=len(all_features),\n        num_cross_layers=4,\n        deep_hidden_dims=[512, 256, 128],\n        use_mixture=use_mixture,\n        num_experts=4,\n        combination='parallel'\n    ).to(device)\n    \n    print(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n    print(f\"Using {'Mixture of Experts' if use_mixture else 'Matrix'} Cross Layers\")\n    \n    # Train model\n    print(\"\\nTraining DCN V2 model...\")\n    criterion = nn.HuberLoss(delta=1.0)\n    optimizer = torch.optim.AdamW(model.parameters(), lr=0.001, weight_decay=1e-4)\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3)\n    \n    best_val_loss = float('inf')\n    patience_counter = 0\n    \n    # Reset gradient importance accumulator\n    model.reset_gradient_importance()\n    \n    for epoch in range(20):  # Train for 20 epochs\n        model.train()\n        train_loss = 0.0\n        \n        for inputs, targets in tqdm(train_loader, desc=f\"Epoch {epoch+1}/20\"):\n            inputs = inputs.to(device)\n            targets = targets.to(device)\n            \n            # Enable gradient computation for inputs\n            inputs.requires_grad = True\n            \n            optimizer.zero_grad()\n            outputs = model(inputs, track_gradients=True)\n            loss = criterion(outputs, targets)\n            \n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n            optimizer.step()\n            \n            train_loss += loss.item()\n        \n        # Validation\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        \n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs, targets = inputs.to(device), targets.to(device)\n                outputs = model(inputs, track_gradients=False)\n                loss = criterion(outputs, targets)\n                val_loss += loss.item()\n                \n                val_preds.extend(outputs.cpu().numpy().flatten())\n                val_targets.extend(targets.cpu().numpy().flatten())\n        \n        avg_train_loss = train_loss / len(train_loader)\n        avg_val_loss = val_loss / len(val_loader)\n        val_pearson = pearsonr(val_targets, val_preds)[0]\n        \n        print(f\"Epoch {epoch+1}: Train Loss = {avg_train_loss:.4f}, Val Loss = {avg_val_loss:.4f}, Val Pearson = {val_pearson:.4f}\")\n        \n        scheduler.step(avg_val_loss)\n        \n        # Early stopping\n        if avg_val_loss < best_val_loss:\n            best_val_loss = avg_val_loss\n            patience_counter = 0\n            torch.save(model.state_dict(), 'best_dcnv2_importance.pt')\n        else:\n            patience_counter += 1\n            if patience_counter >= 5:\n                print(\"Early stopping triggered\")\n                break\n    \n    # Load best model\n    model.load_state_dict(torch.load('best_dcnv2_importance.pt'))\n    \n    # Compute feature importance using multiple methods\n    print(\"\\n=== Computing DCN V2 Feature Importance ===\")\n    \n    # 1. Gradient importance (accumulated during training)\n    print(\"\\n1. Getting accumulated gradient importance...\")\n    gradient_importance = model.get_gradient_importance()\n    \n    # 2. Cross network importance (if using matrix parameterization)\n    print(\"\\n2. Computing cross network importance...\")\n    cross_importance = model.compute_cross_importance()\n    \n    # 3. Feature interactions\n    print(\"\\n3. Computing feature interactions...\")\n    top_features, interaction_matrix, individual_effects = model.compute_feature_interactions(\n        X_val_scaled, n_samples=3000, top_k=20\n    )\n    \n    # 4. Polynomial feature importance\n    print(\"\\n4. Computing polynomial feature importance...\")\n    polynomial_importance = model.compute_polynomial_importance(\n        X_val_scaled, max_order=3, n_samples=2000\n    )\n    \n    # Combine all importance scores\n    importance_results = {\n        'gradient_importance': gradient_importance,\n        'cross_importance': cross_importance,\n        'individual_effects': individual_effects,\n        'interaction_top_features': top_features,\n        'interaction_matrix': interaction_matrix,\n        'polynomial_importance': polynomial_importance,\n        'use_mixture': use_mixture\n    }\n    \n    # Create combined score\n    combined_scores = []\n    \n    # Add gradient importance\n    grad_normalized = (gradient_importance - gradient_importance.min()) / \\\n                     (gradient_importance.max() - gradient_importance.min() + 1e-8)\n    combined_scores.append(grad_normalized)\n    \n    # Add cross importance if available\n    if cross_importance is not None:\n        cross_normalized = (cross_importance - cross_importance.min()) / \\\n                          (cross_importance.max() - cross_importance.min() + 1e-8)\n        combined_scores.append(cross_normalized)\n    \n    # Add individual effects\n    effects_normalized = (individual_effects - individual_effects.min()) / \\\n                        (individual_effects.max() - individual_effects.min() + 1e-8)\n    combined_scores.append(effects_normalized)\n    \n    # Average normalized scores\n    combined_importance = np.mean(combined_scores, axis=0)\n    importance_results['combined_importance'] = combined_importance\n    \n    # Create all visualizations\n    print(\"\\n=== Creating Visualizations ===\")\n    create_dcnv2_visualizations(importance_results, all_features)\n    \n    # Create detailed results dataframe\n    results_df = pd.DataFrame({\n        'feature': all_features,\n        'combined_importance': combined_importance,\n        'gradient_importance': gradient_importance,\n        'individual_effect': individual_effects\n    })\n    \n    # Add cross importance if available\n    if cross_importance is not None:\n        results_df['cross_importance'] = cross_importance\n    \n    # Sort by combined importance\n    results_df = results_df.sort_values('combined_importance', ascending=False)\n    \n    # Display results\n    print(\"\\n\" + \"=\"*80)\n    print(\"DCN V2 FEATURE IMPORTANCE RESULTS\")\n    print(\"=\"*80)\n    \n    print(\"\\n🔍 Top 30 Most Important Features (Combined Score):\")\n    print(\"-\" * 60)\n    for idx, row in results_df.head(30).iterrows():\n        print(f\"{row['feature']:25s} {row['combined_importance']:.6f}\")\n    \n    # Separate by category\n    market_df = results_df[results_df['feature'].isin(market_features)]\n    engineered_df = results_df[results_df['feature'].isin(engineered_features)]\n    x_features_df = results_df[results_df['feature'].str.startswith('X')]\n    \n    print(\"\\n📊 All Market Features Importance:\")\n    print(\"-\" * 60)\n    for idx, row in market_df.iterrows():\n        print(f\"{row['feature']:25s} Combined: {row['combined_importance']:.4f} | \"\n              f\"Gradient: {row['gradient_importance']:.4f} | Effect: {row['individual_effect']:.4f}\")\n    \n    print(\"\\n🔧 Top 20 Engineered Features:\")\n    print(\"-\" * 60)\n    for idx, row in engineered_df.head(20).iterrows():\n        print(f\"{row['feature']:25s} Combined: {row['combined_importance']:.4f} | \"\n              f\"Gradient: {row['gradient_importance']:.4f} | Effect: {row['individual_effect']:.4f}\")\n    \n    print(\"\\n🎯 Top 20 Anonymous Features (X_):\")\n    print(\"-\" * 60)\n    for idx, row in x_features_df.head(20).iterrows():\n        print(f\"{row['feature']:25s} Combined: {row['combined_importance']:.4f} | \"\n              f\"Gradient: {row['gradient_importance']:.4f} | Effect: {row['individual_effect']:.4f}\")\n    \n    # DCN V2-specific insights\n    print(\"\\n🧠 DCN V2-Specific Insights:\")\n    print(\"-\" * 60)\n    print(f\"Cross Network Type: {'Mixture of Experts' if use_mixture else 'Matrix Parameterization'}\")\n    print(f\"Number of Cross Layers: 4\")\n    print(f\"Deep Network Architecture: [512, 256, 128]\")\n    \n    # Top feature interactions\n    print(\"\\n🔗 Top Feature Interactions:\")\n    interaction_pairs = []\n    for i in range(len(top_features)):\n        for j in range(i+1, len(top_features)):\n            if interaction_matrix[i, j] > 0:\n                interaction_pairs.append((\n                    (top_features[i], top_features[j]),\n                    interaction_matrix[i, j]\n                ))\n    \n    interaction_pairs.sort(key=lambda x: x[1], reverse=True)\n    \n    for (feat1, feat2), strength in interaction_pairs[:10]:\n        print(f\"  {all_features[feat1]} × {all_features[feat2]}: {strength:.4f}\")\n    \n    # Top polynomial features\n    print(\"\\n📈 Top Polynomial Features:\")\n    top_polys = sorted(polynomial_importance.items(), key=lambda x: x[1], reverse=True)[:10]\n    for combo, importance in top_polys:\n        if len(combo) == 1:\n            poly_name = all_features[combo[0]]\n        else:\n            poly_name = ' × '.join([all_features[idx] for idx in combo])\n        print(f\"  {poly_name}: {importance:.4f}\")\n    \n    # Save results\n    results_df.to_csv(\"dcnv2_feature_importance_comprehensive.csv\", index=False)\n    print(\"\\n✅ Comprehensive results saved to 'dcnv2_feature_importance_comprehensive.csv'\")\n    \n    # Save interaction analysis\n    interaction_df = pd.DataFrame({\n        'feature1': [all_features[feat1] for (feat1, feat2), _ in interaction_pairs[:50]],\n        'feature2': [all_features[feat2] for (feat1, feat2), _ in interaction_pairs[:50]],\n        'interaction_strength': [strength for _, strength in interaction_pairs[:50]]\n    })\n    interaction_df.to_csv(\"dcnv2_feature_interactions.csv\", index=False)\n    print(\"✅ Top 50 feature interactions saved to 'dcnv2_feature_interactions.csv'\")\n    \n    # Save polynomial features\n    poly_df = pd.DataFrame([\n        {\n            'features': ' × '.join([all_features[idx] for idx in combo]),\n            'order': len(combo),\n            'importance': importance\n        }\n        for combo, importance in sorted(polynomial_importance.items(), \n                                       key=lambda x: x[1], reverse=True)[:100]\n    ])\n    poly_df.to_csv(\"dcnv2_polynomial_features.csv\", index=False)\n    print(\"✅ Top 100 polynomial features saved to 'dcnv2_polynomial_features.csv'\")\n    \n    # Save detailed analysis\n    with open('dcnv2_feature_analysis.txt', 'w') as f:\n        f.write(\"DCN V2 Feature Importance Analysis\\n\")\n        f.write(\"=\"*80 + \"\\n\\n\")\n        \n        f.write(\"Model Configuration:\\n\")\n        f.write(f\"- Cross Network: {'Mixture of Experts' if use_mixture else 'Matrix'}\\n\")\n        f.write(f\"- Number of Cross Layers: 4\\n\")\n        f.write(f\"- Deep Network: [512, 256, 128]\\n\")\n        f.write(f\"- Combination: Parallel\\n\\n\")\n        \n        # Top features by each method\n        methods = ['gradient_importance', 'individual_effect']\n        if cross_importance is not None:\n            methods.append('cross_importance')\n            \n        for method in methods:\n            f.write(f\"\\nTop 20 Features by {method}:\\n\")\n            f.write(\"-\"*60 + \"\\n\")\n            top_by_method = results_df.nlargest(20, method)[['feature', method]]\n            for idx, row in top_by_method.iterrows():\n                f.write(f\"{row['feature']:25s} {row[method]:.6f}\\n\")\n    \n    print(\"✅ Detailed analysis saved to 'dcnv2_feature_analysis.txt'\")\n    \n    return results_df, importance_results\n\nif __name__ == \"__main__\":\n    results_df, importance_dict = analyze_dcnv2_feature_importance()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}