{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# NODE Implementation for Crypto Market Price Prediction\n# Neural Oblivious Decision Ensembles for enhanced interpretability and performance\n\n# Import packages\nimport random\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, QuantileTransformer\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.decomposition import PCA\nfrom tqdm import tqdm\nimport gc\nfrom typing import List, Tuple, Dict, Optional\nimport math\nimport pickle\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau, CosineAnnealingWarmRestarts\nimport torch.nn.functional as F\n\nfrom scipy.stats import pearsonr, spearmanr\nfrom scipy.special import erfinv\nfrom scipy.stats import kurtosis, skew\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Feature Configuration (unchanged)\n# =========================\nMARKET_FEATURES = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n\nPROPRIETARY_FEATURES = [\n    \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n    \"X425\", \"X132\", \"X691\", \"X593\", \"X377\", \"X285\", \"X126\", \"X419\", \"X604\",\n    \"X84\", \"X138\", \"X413\", \"X291\", \"X40\", \"X123\", \"X81\", \"X853\", \"X854\",\n    \"X777\", \"X219\", \"X776\", \"X180\", \"X781\", \"X445\", \"X444\", \"X384\", \"X466\",\n    \"X95\", \"X583\", \"X272\", \"X137\", \"X533\", \"X758\", \"X279\", \"X297\",\n    \"X21\", \"X20\", \"X28\", \"X29\", \"X19\", \"X27\", \"X22\", \"X198\", \"X89\", \"X90\",\n    \"X98\", \"X96\", \"X97\", \"X383\", \"X427\", \"X451\", \"X283\",\n    \"X753\", \"X497\", \"X748\", \"X820\", \"X566\", \"X535\", \"X394\", \"X618\",\n    \"X429\", \"X381\", \"X387\", \"X890\", \"X752\", \"X375\", \"X68\", \"X152\",\n    \"X110\", \"X850\", \"X851\", \"X481\", \"X321\", \"X363\", \"X405\", \"X492\",\n    \"X888\", \"X421\", \"X333\", \"X817\", \"X586\", \"X292\", \"X344\", \"X532\"\n]\n\nSELECTED_FEATURES = MARKET_FEATURES + PROPRIETARY_FEATURES + [\"label\"]\n\n# =========================\n# Feature Engineering (keep existing functions)\n# =========================\ndef add_advanced_market_features(df):\n    \"\"\"Add comprehensive market microstructure features for crypto markets\"\"\"\n    \n    # Core microstructure features\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Advanced volume analysis\n    df['log_volume'] = np.log1p(df['volume'])\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['volume_intensity'] = df['volume'] / (df['volume'].rolling(window=10, min_periods=1).mean() + 1e-10)\n    \n    # Liquidity metrics\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['depth_exhaustion'] = df['volume'] / (df['total_depth'] + 1e-10)\n    \n    # Price pressure and toxicity indicators\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['amihud_illiquidity'] = np.abs(df['order_flow_imbalance']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * np.log1p(df['volume'])\n    \n    # Market stress and volatility proxies\n    df['volume_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['depth_volume_interaction'] = df['total_depth'] * np.log1p(df['volume'])\n    df['flow_intensity'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['volatility_proxy'] = np.abs(df['net_order_flow']) / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Order book shape metrics\n    df['book_pressure'] = (df['bid_qty'] - df['ask_qty']) / (df['volume'] + 1e-10)\n    df['book_skew'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10) * np.sign(df['net_order_flow'])\n    df['depth_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-10)\n    df['inverse_depth_ratio'] = df['ask_qty'] / (df['bid_qty'] + 1e-10)\n    \n    # Execution quality metrics\n    df['execution_probability'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['fill_rate'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Information asymmetry measures\n    df['pin_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['information_share'] = df['net_order_flow'] / (df['volume'] + 1e-10) * np.abs(df['depth_imbalance'])\n    df['adverse_selection'] = np.abs(df['net_order_flow']) / (df['bid_qty'] + df['ask_qty'] + 1e-10) * df['volume']\n    \n    # Complex interaction features\n    df['volume_adjusted_imbalance'] = df['order_flow_imbalance'] * np.log1p(df['volume'])\n    df['depth_adjusted_flow'] = df['net_order_flow'] / (np.log1p(df['total_depth']) + 1)\n    df['liquidity_weighted_pressure'] = df['market_stress'] * df['liquidity_ratio']\n    df['toxic_volume_ratio'] = df['flow_toxicity'] / (df['volume'] + 1e-10)\n    \n    # Non-linear transformations\n    df['volume_squared'] = df['volume'] ** 2\n    df['volume_cubed'] = df['volume'] ** 3\n    df['log_total_activity'] = np.log1p(df['buy_qty'] + df['sell_qty'] + df['volume'])\n    df['sqrt_market_activity'] = np.sqrt(df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'])\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df\n\ndef compute_rolling_features(df, windows=[5, 10, 20]):\n    \"\"\"Compute rolling statistics for key features\"\"\"\n    key_features = ['volume', 'net_order_flow', 'total_depth', 'order_flow_imbalance']\n    \n    for feature in key_features:\n        if feature in df.columns:\n            for window in windows:\n                df[f'{feature}_ma_{window}'] = df[feature].rolling(window=window, min_periods=1).mean()\n                df[f'{feature}_std_{window}'] = df[feature].rolling(window=window, min_periods=1).std().fillna(0)\n                df[f'{feature}_skew_{window}'] = df[feature].rolling(window=window, min_periods=3).skew().fillna(0)\n                df[f'{feature}_rel_ma_{window}'] = df[feature] / (df[f'{feature}_ma_{window}'] + 1e-10)\n    \n    return df\n\n# =========================\n# Data Transformation Functions (keep existing)\n# =========================\ndef apply_advanced_transformations(X, method='quantile'):\n    \"\"\"Apply advanced transformations to features\"\"\"\n    if method == 'quantile':\n        transformer = QuantileTransformer(output_distribution='normal', random_state=42)\n    elif method == 'robust':\n        transformer = RobustScaler()\n    elif method == 'standard':\n        transformer = StandardScaler()\n    else:\n        raise ValueError(f\"Unknown transformation method: {method}\")\n    \n    return transformer.fit_transform(X), transformer\n\ndef add_noise_augmentation(X, noise_factor=0.01):\n    \"\"\"Apply Gaussian noise for regularization\"\"\"\n    if np.random.random() < 0.8:\n        noise = np.random.normal(0, noise_factor, X.shape)\n        return X + noise * np.std(X, axis=0)\n    return X\n\ndef compute_comprehensive_feature_importance(X, y, feature_names, n_samples=50000, n_permutations=10):\n    \"\"\"Compute feature importance using multiple methods\"\"\"\n    print(f\"Computing feature importance with {n_samples} samples...\")\n    \n    if len(X) <= n_samples:\n        X_subset = X\n        y_subset = y\n    else:\n        indices = np.random.choice(len(X), n_samples, replace=False)\n        X_subset = X[indices]\n        y_subset = y.iloc[indices] if hasattr(y, 'iloc') else y[indices]\n    \n    mi_scores = mutual_info_regression(X_subset, y_subset, random_state=42, n_neighbors=5)\n    \n    baseline_corr = pearsonr(y_subset, np.mean(X_subset, axis=1))[0]\n    \n    perm_importance = np.zeros(X_subset.shape[1])\n    for i in range(X_subset.shape[1]):\n        perm_scores = []\n        for _ in range(n_permutations):\n            X_perm = X_subset.copy()\n            np.random.shuffle(X_perm[:, i])\n            perm_corr = pearsonr(y_subset, np.mean(X_perm, axis=1))[0]\n            perm_scores.append(baseline_corr - perm_corr)\n        perm_importance[i] = np.mean(perm_scores)\n    \n    combined_scores = (mi_scores / mi_scores.max() + \n                      perm_importance / (perm_importance.max() + 1e-10)) / 2\n    \n    importance_df = pd.DataFrame({\n        'feature': feature_names,\n        'mi_score': mi_scores,\n        'perm_importance': perm_importance,\n        'combined_score': combined_scores\n    }).sort_values('combined_score', ascending=False)\n    \n    return importance_df\n\n# =========================\n# NODE Architecture Components\n# =========================\ndef set_seed(seed=42):\n    \"\"\"Set random seeds for reproducibility\"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nclass ObliviousDecisionTree(nn.Module):\n    \"\"\"Differentiable Oblivious Decision Tree for NODE\"\"\"\n    def __init__(self, depth, num_features, temperature=1.0):\n        super().__init__()\n        self.depth = depth\n        self.num_features = num_features\n        self.temperature = temperature\n        self.num_leaves = 2 ** depth\n        \n        # Initialize split features and thresholds\n        self.split_features = nn.Parameter(torch.randn(depth, num_features))\n        self.split_thresholds = nn.Parameter(torch.zeros(depth))\n        \n        # Initialize leaf predictions\n        self.leaf_values = nn.Parameter(torch.randn(self.num_leaves))\n        \n        # Temperature parameter for soft routing\n        self.temperature_param = nn.Parameter(torch.tensor(temperature))\n        \n    def forward(self, x):\n        batch_size = x.size(0)\n        \n        # Compute routing decisions for all depths\n        routing_decisions = []\n        for d in range(self.depth):\n            # Feature selection using soft attention\n            feature_weights = F.softmax(self.split_features[d] / self.temperature_param, dim=0)\n            selected_feature = torch.sum(x * feature_weights.unsqueeze(0), dim=1)\n            \n            # Soft routing decision\n            decision = torch.sigmoid((selected_feature - self.split_thresholds[d]) / self.temperature_param)\n            routing_decisions.append(decision)\n        \n        # Compute leaf indices using soft routing\n        leaf_probs = torch.ones(batch_size, self.num_leaves, device=x.device)\n        \n        for d in range(self.depth):\n            decision = routing_decisions[d].unsqueeze(1)\n            \n            # Update probabilities for left and right subtrees\n            left_mask = torch.zeros(batch_size, self.num_leaves, device=x.device)\n            right_mask = torch.zeros(batch_size, self.num_leaves, device=x.device)\n            \n            for leaf_idx in range(self.num_leaves):\n                if (leaf_idx >> (self.depth - d - 1)) & 1 == 0:\n                    left_mask[:, leaf_idx] = 1\n                else:\n                    right_mask[:, leaf_idx] = 1\n            \n            leaf_probs = leaf_probs * (left_mask * (1 - decision) + right_mask * decision)\n        \n        # Compute final predictions\n        predictions = torch.sum(leaf_probs * self.leaf_values.unsqueeze(0), dim=1)\n        \n        return predictions, leaf_probs\n\nclass NODE(nn.Module):\n    \"\"\"Neural Oblivious Decision Ensembles\"\"\"\n    def __init__(self, config):\n        super().__init__()\n        \n        self.num_features = config['num_features']\n        self.num_trees = config['num_trees']\n        self.tree_depth = config['tree_depth']\n        self.temperature = config['temperature']\n        self.use_feature_embedding = config.get('use_feature_embedding', True)\n        \n        # Feature embedding layer (optional)\n        if self.use_feature_embedding:\n            self.feature_embedder = nn.Sequential(\n                nn.Linear(self.num_features, config['embedding_dim']),\n                nn.LayerNorm(config['embedding_dim']),\n                nn.GELU(),\n                nn.Dropout(config['dropout']),\n                nn.Linear(config['embedding_dim'], self.num_features),\n                nn.LayerNorm(self.num_features)\n            )\n        \n        # Create ensemble of oblivious decision trees\n        self.trees = nn.ModuleList([\n            ObliviousDecisionTree(\n                depth=self.tree_depth,\n                num_features=self.num_features,\n                temperature=self.temperature\n            ) for _ in range(self.num_trees)\n        ])\n        \n        # Tree weighting mechanism\n        self.tree_weights = nn.Parameter(torch.ones(self.num_trees))\n        \n        # Optional deep layers for final prediction\n        if config.get('use_deep_layers', True):\n            self.deep_layers = nn.Sequential(\n                nn.Linear(self.num_trees, config['hidden_dim']),\n                nn.LayerNorm(config['hidden_dim']),\n                nn.GELU(),\n                nn.Dropout(config['dropout']),\n                nn.Linear(config['hidden_dim'], config['hidden_dim'] // 2),\n                nn.LayerNorm(config['hidden_dim'] // 2),\n                nn.GELU(),\n                nn.Dropout(config['dropout']),\n                nn.Linear(config['hidden_dim'] // 2, 1)\n            )\n        else:\n            self.deep_layers = None\n        \n        self._initialize_weights()\n    \n    def _initialize_weights(self):\n        for module in self.modules():\n            if isinstance(module, nn.Linear):\n                nn.init.xavier_uniform_(module.weight)\n                if module.bias is not None:\n                    nn.init.constant_(module.bias, 0)\n    \n    def forward(self, x):\n        # Apply feature embedding if enabled\n        if self.use_feature_embedding:\n            x_embedded = self.feature_embedder(x) + x  # Residual connection\n        else:\n            x_embedded = x\n        \n        # Get predictions from all trees\n        tree_outputs = []\n        tree_leaf_probs = []\n        \n        for tree in self.trees:\n            output, leaf_probs = tree(x_embedded)\n            tree_outputs.append(output)\n            tree_leaf_probs.append(leaf_probs)\n        \n        tree_outputs = torch.stack(tree_outputs, dim=1)\n        \n        # Apply tree weights\n        tree_weights_normalized = F.softmax(self.tree_weights, dim=0)\n        \n        if self.deep_layers is not None:\n            # Use deep layers for final prediction\n            final_output = self.deep_layers(tree_outputs)\n        else:\n            # Weighted average of tree outputs\n            final_output = torch.sum(tree_outputs * tree_weights_normalized.unsqueeze(0), dim=1, keepdim=True)\n        \n        return final_output, tree_outputs, tree_leaf_probs\n\nclass NODEEnsemble(nn.Module):\n    \"\"\"Ensemble of NODE models with different configurations\"\"\"\n    def __init__(self, base_config, num_models=3):\n        super().__init__()\n        \n        self.num_models = num_models\n        self.models = nn.ModuleList()\n        \n        # Create models with varying configurations\n        for i in range(num_models):\n            model_config = base_config.copy()\n            \n            # Vary tree depth and number of trees\n            if i == 0:\n                model_config['tree_depth'] = 6\n                model_config['num_trees'] = 100\n            elif i == 1:\n                model_config['tree_depth'] = 5\n                model_config['num_trees'] = 150\n            else:\n                model_config['tree_depth'] = 7\n                model_config['num_trees'] = 80\n            \n            self.models.append(NODE(model_config))\n        \n        # Ensemble weighting\n        self.ensemble_weights = nn.Parameter(torch.ones(num_models))\n        \n    def forward(self, x):\n        outputs = []\n        all_tree_outputs = []\n        all_leaf_probs = []\n        \n        for model in self.models:\n            output, tree_outputs, leaf_probs = model(x)\n            outputs.append(output)\n            all_tree_outputs.append(tree_outputs)\n            all_leaf_probs.append(leaf_probs)\n        \n        outputs = torch.stack(outputs, dim=1)\n        weights = F.softmax(self.ensemble_weights, dim=0)\n        \n        final_output = torch.sum(outputs * weights.view(1, -1, 1), dim=1)\n        \n        return final_output, outputs, all_tree_outputs, all_leaf_probs\n\n# =========================\n# Training Functions for NODE\n# =========================\ndef train_node(model, train_loader, val_loader, config, device):\n    \"\"\"Train NODE with advanced optimization strategies\"\"\"\n    \n    # Loss function\n    criterion = nn.HuberLoss(delta=config['huber_delta'])\n    \n    # Optimizer with different learning rates for different components\n    param_groups = [\n        {'params': [p for n, p in model.named_parameters() if 'tree' in n], \n         'lr': config['tree_lr']},\n        {'params': [p for n, p in model.named_parameters() if 'tree' not in n], \n         'lr': config['learning_rate']}\n    ]\n    \n    optimizer = optim.AdamW(param_groups, weight_decay=config['weight_decay'], eps=1e-8)\n    \n    # Learning rate scheduler\n    scheduler = ReduceLROnPlateau(\n        optimizer, \n        mode='max', \n        factor=0.5, \n        patience=5, \n        verbose=True,\n        min_lr=1e-7\n    )\n    \n    # Temperature annealing schedule\n    temperature_schedule = lambda epoch: max(0.1, config['temperature'] * (0.9 ** epoch))\n    \n    # Training history\n    history = {'train_loss': [], 'val_loss': [], 'val_pearson': []}\n    best_pearson = -np.inf\n    patience_counter = 0\n    \n    for epoch in range(config['num_epochs']):\n        # Update temperature for all trees\n        current_temperature = temperature_schedule(epoch)\n        for tree_module in model.modules():\n            if isinstance(tree_module, ObliviousDecisionTree):\n                tree_module.temperature_param.data.fill_(current_temperature)\n        \n        # Training phase\n        model.train()\n        train_loss = 0.0\n        train_batches = 0\n        \n        progress_bar = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{config['num_epochs']}\")\n        \n        for inputs, targets in progress_bar:\n            inputs, targets = inputs.to(device), targets.to(device)\n            \n            # Apply noise augmentation\n            if config.get('use_noise_augmentation', True):\n                inputs = inputs + torch.randn_like(inputs) * config['noise_factor']\n            \n            optimizer.zero_grad()\n            \n            # Forward pass\n            if isinstance(model, NODEEnsemble):\n                outputs, _, _, _ = model(inputs)\n            else:\n                outputs, _, _ = model(inputs)\n            \n            loss = criterion(outputs, targets)\n            \n            # Add regularization\n            if config.get('l2_penalty', 0) > 0:\n                l2_reg = config['l2_penalty'] * sum(p.pow(2.0).sum() for p in model.parameters())\n                loss = loss + l2_reg\n            \n            # Add entropy regularization for tree routing\n            if config.get('entropy_reg', 0) > 0 and hasattr(model, 'trees'):\n                entropy_loss = 0\n                for tree in model.trees if hasattr(model, 'trees') else [model]:\n                    if isinstance(tree, ObliviousDecisionTree):\n                        # Compute entropy of split decisions\n                        split_probs = torch.sigmoid(tree.split_features / current_temperature)\n                        entropy = -torch.sum(split_probs * torch.log(split_probs + 1e-10) + \n                                           (1 - split_probs) * torch.log(1 - split_probs + 1e-10))\n                        entropy_loss += entropy\n                loss = loss - config['entropy_reg'] * entropy_loss / len(model.trees if hasattr(model, 'trees') else [model])\n            \n            # Backward pass\n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=config['grad_clip'])\n            optimizer.step()\n            \n            train_loss += loss.item()\n            train_batches += 1\n            \n            progress_bar.set_postfix({'loss': f'{loss.item():.4f}', 'temp': f'{current_temperature:.3f}'})\n        \n        avg_train_loss = train_loss / train_batches\n        history['train_loss'].append(avg_train_loss)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        \n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs, targets = inputs.to(device), targets.to(device)\n                \n                if isinstance(model, NODEEnsemble):\n                    outputs, _, _, _ = model(inputs)\n                else:\n                    outputs, _, _ = model(inputs)\n                    \n                loss = criterion(outputs, targets)\n                \n                val_loss += loss.item()\n                val_preds.extend(outputs.cpu().numpy().flatten())\n                val_targets.extend(targets.cpu().numpy().flatten())\n        \n        avg_val_loss = val_loss / len(val_loader)\n        val_pearson = pearsonr(val_targets, val_preds)[0]\n        val_spearman = spearmanr(val_targets, val_preds)[0]\n        \n        history['val_loss'].append(avg_val_loss)\n        history['val_pearson'].append(val_pearson)\n        \n        # Update scheduler\n        scheduler.step(val_pearson)\n        \n        print(f\"\\nEpoch {epoch+1}: Train Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")\n        print(f\"Val Pearson: {val_pearson:.4f}, Val Spearman: {val_spearman:.4f}\")\n        print(f\"Temperature: {current_temperature:.3f}, LR: {optimizer.param_groups[0]['lr']:.6f}\")\n        \n        # Save best model\n        if val_pearson > best_pearson:\n            best_pearson = val_pearson\n            torch.save({\n                'model_state_dict': model.state_dict(),\n                'config': config,\n                'epoch': epoch,\n                'best_pearson': best_pearson\n            }, f\"best_node_{config.get('model_id', 'default')}.pt\")\n            print(f\"✅ New best model saved with Pearson: {val_pearson:.4f}\")\n            patience_counter = 0\n        else:\n            patience_counter += 1\n            if patience_counter >= config['patience']:\n                print(f\"Early stopping triggered after {epoch+1} epochs\")\n                break\n    \n    # Load best model with fixed weights_only parameter\n    checkpoint = torch.load(\n        f\"best_node_{config.get('model_id', 'default')}.pt\",\n        weights_only=False,\n        map_location=device\n    )\n    model.load_state_dict(checkpoint['model_state_dict'])\n    \n    return model, history, best_pearson\n\n# =========================\n# Main Execution with NODE\n# =========================\ndef main():\n    print(\"=== NODE (Neural Oblivious Decision Ensembles) for Crypto Market Prediction ===\")\n    \n    # Load training data\n    print(\"\\nLoading training data...\")\n    train = pl.scan_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\n    train = train.select(SELECTED_FEATURES).collect().to_pandas()\n    \n    print(f\"Initial training data shape: {train.shape}\")\n    \n    # Use recent data for better relevance\n    train_size = int(0.85 * len(train))\n    train = train.iloc[-train_size:].reset_index(drop=True)\n    print(f\"Using last {train_size} samples for training\")\n    \n    # Add market microstructure features\n    print(\"\\nAdding advanced market microstructure features...\")\n    train = add_advanced_market_features(train)\n    train = compute_rolling_features(train)\n    \n    # Get all feature names\n    all_features = [col for col in train.columns if col != 'label']\n    print(f\"Total number of features after engineering: {len(all_features)}\")\n    \n    # Split data\n    X_train, X_val = train_test_split(train, test_size=0.2, shuffle=False, random_state=42)\n    \n    y_train = X_train.pop(\"label\")\n    y_val = X_val.pop(\"label\")\n    \n    # Feature importance analysis\n    print(\"\\nComputing comprehensive feature importance...\")\n    importance_df = compute_comprehensive_feature_importance(\n        X_train[all_features].values,\n        y_train,\n        all_features,\n        n_samples=min(100000, len(X_train))\n    )\n    \n    # Select features with importance threshold\n    feature_threshold = 0.01\n    selected_features = importance_df[importance_df['combined_score'] > feature_threshold]['feature'].tolist()\n    print(f\"Features with importance > {feature_threshold}: {len(selected_features)}\")\n    \n    # Prepare data with selected features\n    X_train_selected = X_train[selected_features].values\n    X_val_selected = X_val[selected_features].values\n    \n    # Apply transformations\n    print(\"\\nApplying data transformations...\")\n    X_train_transformed, transformer = apply_advanced_transformations(X_train_selected, method='quantile')\n    X_val_transformed = transformer.transform(X_val_selected)\n    \n    print(f\"Final training shape: {X_train_transformed.shape}\")\n    print(f\"Final validation shape: {X_val_transformed.shape}\")\n    \n    # NODE configuration\n    base_config = {\n        'num_features': X_train_transformed.shape[1],\n        'num_trees': 100,\n        'tree_depth': 6,\n        'temperature': 1.0,\n        'use_feature_embedding': True,\n        'embedding_dim': 128,\n        'use_deep_layers': True,\n        'hidden_dim': 256,\n        'dropout': 0.2,\n        'num_epochs': 50,\n        'batch_size': 512,\n        'learning_rate': 0.001,\n        'tree_lr': 0.0005,\n        'weight_decay': 0.01,\n        'huber_delta': 1.0,\n        'noise_factor': 0.01,\n        'grad_clip': 1.0,\n        'patience': 10,\n        'l2_penalty': 0.0001,\n        'entropy_reg': 0.001\n    }\n    \n    # Create data loaders\n    train_dataset = TensorDataset(\n        torch.tensor(X_train_transformed, dtype=torch.float32),\n        torch.tensor(y_train.values, dtype=torch.float32).unsqueeze(1)\n    )\n    val_dataset = TensorDataset(\n        torch.tensor(X_val_transformed, dtype=torch.float32),\n        torch.tensor(y_val.values, dtype=torch.float32).unsqueeze(1)\n    )\n    \n    train_loader = DataLoader(train_dataset, batch_size=base_config['batch_size'], \n                            shuffle=True, num_workers=0)\n    val_loader = DataLoader(val_dataset, batch_size=base_config['batch_size'], \n                          shuffle=False, num_workers=0)\n    \n    # Train ensemble of NODE models\n    print(\"\\n=== Training Ensemble of NODE Models ===\")\n    \n    ensemble_models = []\n    model_configs = [\n        {'model_id': 'node_v1', 'seed': 42},\n        {'model_id': 'node_v2', 'seed': 1337, 'num_trees': 120, 'tree_depth': 5},\n        {'model_id': 'node_v3', 'seed': 2468, 'num_trees': 80, 'tree_depth': 7}\n    ]\n    \n    for i, config_update in enumerate(model_configs[:2]):  # Use 2 models for ensemble\n        print(f\"\\n--- Training NODE Model {i+1} ---\")\n        \n        # Update configuration\n        model_config = base_config.copy()\n        model_config.update(config_update)\n        \n        set_seed(model_config['seed'])\n        \n        # Create model\n        if i == 0:\n            # Train single NODE model\n            model = NODE(model_config).to(device)\n        else:\n            # Train NODE ensemble\n            model = NODEEnsemble(model_config, num_models=3).to(device)\n        \n        print(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n        \n        # Train model\n        model, history, best_pearson = train_node(\n            model, train_loader, val_loader, model_config, device\n        )\n        \n        print(f\"Model {i+1} best validation Pearson: {best_pearson:.4f}\")\n        \n        # Store model for ensemble\n        ensemble_models.append((model, transformer, selected_features))\n    \n    # Test prediction\n    print(\"\\n=== Making Test Predictions ===\")\n    \n    # Load test data\n    test = pl.scan_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\n    test_features = [f for f in SELECTED_FEATURES if f != \"label\"]\n    test = test.select(test_features).collect().to_pandas()\n    \n    # Add features to test data\n    test = add_advanced_market_features(test)\n    test = compute_rolling_features(test)\n    \n    # Make ensemble predictions\n    all_predictions = []\n    \n    for model, transformer, features in ensemble_models:\n        # Prepare test data\n        X_test_selected = test[features].values\n        X_test_transformed = transformer.transform(X_test_selected)\n        \n        # Make predictions\n        model.eval()\n        test_dataset = TensorDataset(torch.tensor(X_test_transformed, dtype=torch.float32))\n        test_loader = DataLoader(test_dataset, batch_size=2048, shuffle=False)\n        \n        predictions = []\n        with torch.no_grad():\n            for (inputs,) in test_loader:\n                inputs = inputs.to(device)\n                \n                if isinstance(model, NODEEnsemble):\n                    outputs, _, _, _ = model(inputs)\n                else:\n                    outputs, _, _ = model(inputs)\n                    \n                predictions.extend(outputs.cpu().numpy().flatten())\n        \n        all_predictions.append(np.array(predictions))\n    \n    # Ensemble predictions\n    final_predictions = np.mean(all_predictions, axis=0)\n    \n    # Post-processing\n    pred_mean = y_train.mean()\n    pred_std = y_train.std()\n    final_predictions = np.clip(final_predictions, \n                               pred_mean - 4 * pred_std, \n                               pred_mean + 4 * pred_std)\n    \n    # Create submission\n    submission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\n    submission[\"prediction\"] = final_predictions\n    submission.to_csv(\"submission_node.csv\", index=False)\n    \n    # Display results\n    print(\"\\n=== Final Results ===\")\n    print(f\"Ensemble size: {len(ensemble_models)} NODE models\")\n    print(f\"Features used: {len(selected_features)}\")\n    print(f\"\\nPrediction statistics:\")\n    print(f\"Mean: {final_predictions.mean():.6f}\")\n    print(f\"Std: {final_predictions.std():.6f}\")\n    print(f\"Min: {final_predictions.min():.6f}\")\n    print(f\"Max: {final_predictions.max():.6f}\")\n    print(f\"Skewness: {skew(final_predictions):.4f}\")\n    print(f\"Kurtosis: {kurtosis(final_predictions):.4f}\")\n    \n    print(\"\\nTop 15 most important features:\")\n    print(importance_df.head(15)[['feature', 'combined_score']])\n    \n    # Model interpretability analysis\n    print(\"\\n=== NODE Model Interpretability ===\")\n    print(\"NODE provides interpretability through its oblivious decision tree structure.\")\n    print(\"Each tree uses the same splitting features at each depth level across all samples.\")\n    print(\"This makes the model more interpretable than traditional neural networks\")\n    print(\"while maintaining strong predictive performance.\")\n    \n    print(\"\\nSubmission saved as 'submission_node.csv'\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}