{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python\n# -*- coding: utf-8 -*-\n\"\"\"\nDRW Crypto Market Prediction - Enhanced with Genetic Programming Features\nCombines original model with GP-evolved feature engineering\n\"\"\"\n\nimport sys\nimport os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nfrom scipy.special import expit\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\n\nprint(\"=\" * 80)\nprint(\"DRW CRYPTO PREDICTION - GENETIC PROGRAMMING ENHANCED\")\nprint(\"=\" * 80)\n\n# ==============================================================================\n# PART 1: GENETIC PROGRAMMING FEATURE GENERATOR\n# ==============================================================================\n\nclass GeneticProgrammingFeatures:\n    \"\"\"\n    Generate features using genetic programming principles.\n    Includes unary operators, binary operators, and complex combinations.\n    \"\"\"\n    \n    def __init__(self):\n        # Define genetic programming operators\n        self.unary_ops = {\n            'sin': np.sin,\n            'cos': np.cos,\n            'tan': lambda x: np.tanh(x),  # Use tanh instead of tan for stability\n            'exp': lambda x: np.exp(np.clip(x, -5, 5)),\n            'log': lambda x: np.log(np.abs(x) + 1e-8),\n            'sqrt': lambda x: np.sqrt(np.abs(x)),\n            'square': lambda x: x ** 2,\n            'cube': lambda x: x ** 3,\n            'inv': lambda x: 1 / (x + 1e-8),\n            'sigmoid': lambda x: 1 / (1 + np.exp(-np.clip(x, -10, 10))),\n            'relu': lambda x: np.maximum(0, x),\n            'sign': np.sign\n        }\n        \n        self.binary_ops = {\n            'add': lambda x, y: x + y,\n            'sub': lambda x, y: x - y,\n            'mul': lambda x, y: x * y,\n            'div': lambda x, y: x / (y + 1e-8),\n            'pow': lambda x, y: np.abs(x) ** np.clip(y, -2, 2),\n            'max': np.maximum,\n            'min': np.minimum,\n            'mod': lambda x, y: x % (np.abs(y) + 1e-8)\n        }\n        \n    def safe_normalize(self, x):\n        \"\"\"Safely normalize features to prevent numerical issues.\"\"\"\n        std = np.std(x)\n        if std < 1e-8:\n            return np.zeros_like(x)\n        return (x - np.mean(x)) / std\n    \n    def create_gp_features(self, df, top_features):\n        \"\"\"\n        Create GP-evolved features based on top performing base features.\n        \"\"\"\n        print(\"\\nGenerating Genetic Programming features...\")\n        \n        gp_features = []\n        feature_names = []\n        \n        # 1. UNARY TRANSFORMATIONS ON TOP FEATURES\n        for i, feat in enumerate(top_features[:5]):  # Top 5 features\n            if feat in df.columns:\n                x = self.safe_normalize(df[feat].values)\n                \n                # Apply multiple unary operators\n                gp_features.append(np.sin(x) * np.cos(x))\n                feature_names.append(f'gp_sincos_{feat}')\n                \n                gp_features.append(np.log(np.abs(x) + 1) * np.sign(x))\n                feature_names.append(f'gp_logsign_{feat}')\n                \n                gp_features.append(np.tanh(x) * np.exp(-np.abs(x)))\n                feature_names.append(f'gp_tanhexp_{feat}')\n        \n        # 2. BINARY COMBINATIONS OF TOP FEATURES\n        for i in range(min(3, len(top_features))):\n            for j in range(i + 1, min(5, len(top_features))):\n                if top_features[i] in df.columns and top_features[j] in df.columns:\n                    x = self.safe_normalize(df[top_features[i]].values)\n                    y = self.safe_normalize(df[top_features[j]].values)\n                    \n                    # Complex binary operations\n                    gp_features.append(np.sin(x) * np.cos(y))\n                    feature_names.append(f'gp_sincos_{top_features[i]}_{top_features[j]}')\n                    \n                    gp_features.append(x * y / (x**2 + y**2 + 1e-8))\n                    feature_names.append(f'gp_ratio_{top_features[i]}_{top_features[j]}')\n                    \n                    gp_features.append(np.sqrt(np.abs(x * y)) * np.sign(x - y))\n                    feature_names.append(f'gp_sqrtsign_{top_features[i]}_{top_features[j]}')\n        \n        # 3. MARKET MICROSTRUCTURE GP FEATURES\n        if all(col in df.columns for col in ['bid_qty', 'ask_qty', 'volume', 'buy_qty', 'sell_qty']):\n            bid = df['bid_qty'].values\n            ask = df['ask_qty'].values\n            vol = df['volume'].values\n            buy = df['buy_qty'].values\n            sell = df['sell_qty'].values\n            \n            # GP Feature: Order flow oscillator\n            flow_ratio = (buy - sell) / (buy + sell + 1e-8)\n            oscillator = np.sin(flow_ratio * np.pi) * np.log1p(vol)\n            gp_features.append(oscillator)\n            feature_names.append('gp_flow_oscillator')\n            \n            # GP Feature: Pressure dynamics\n            pressure = (bid - ask) / (bid + ask + 1e-8)\n            dynamics = np.tanh(pressure) * np.sqrt(vol / (vol.mean() + 1e-8))\n            gp_features.append(dynamics)\n            feature_names.append('gp_pressure_dynamics')\n            \n            # GP Feature: Volume-weighted imbalance\n            imbalance = np.sign(buy - sell) * np.log1p(np.abs(buy - sell))\n            vol_weight = np.exp(-vol / (vol.std() + 1e-8))\n            gp_features.append(imbalance * vol_weight)\n            feature_names.append('gp_vol_weighted_imbalance')\n            \n            # GP Feature: Market stress indicator\n            spread = np.abs(bid - ask) / (bid + ask + 1e-8)\n            stress = spread * np.exp(flow_ratio) / (1 + np.log1p(vol))\n            gp_features.append(stress)\n            feature_names.append('gp_market_stress')\n        \n        # 4. COMPLEX TERNARY INTERACTIONS\n        if len(top_features) >= 3:\n            for i in range(min(2, len(top_features) - 2)):\n                feat1, feat2, feat3 = top_features[i:i+3]\n                if all(f in df.columns for f in [feat1, feat2, feat3]):\n                    x = self.safe_normalize(df[feat1].values)\n                    y = self.safe_normalize(df[feat2].values)\n                    z = self.safe_normalize(df[feat3].values)\n                    \n                    # GP Feature: Three-way cyclic interaction\n                    cyclic = np.sin(x) * np.cos(y) * np.tanh(z)\n                    gp_features.append(cyclic)\n                    feature_names.append(f'gp_cyclic_{feat1}_{feat2}_{feat3}')\n                    \n                    # GP Feature: Conditional interaction\n                    conditional = np.where(z > 0, x * y, x / (y + 1e-8))\n                    gp_features.append(conditional)\n                    feature_names.append(f'gp_conditional_{feat1}_{feat2}_{feat3}')\n        \n        # 5. TIME-SERIES INSPIRED GP FEATURES\n        if len(top_features) >= 2:\n            for feat in top_features[:3]:\n                if feat in df.columns:\n                    x = df[feat].values\n                    \n                    # GP Feature: Momentum-like\n                    momentum = np.concatenate([[0], np.diff(x)])\n                    momentum_transform = np.tanh(momentum) * np.sqrt(np.abs(x))\n                    gp_features.append(momentum_transform)\n                    feature_names.append(f'gp_momentum_{feat}')\n                    \n                    # GP Feature: Volatility-like\n                    rolling_std = pd.Series(x).rolling(10, min_periods=1).std().fillna(0).values\n                    vol_transform = np.log1p(rolling_std) * np.sign(x)\n                    gp_features.append(vol_transform)\n                    feature_names.append(f'gp_volatility_{feat}')\n        \n        # 6. DISTANCE AND KERNEL GP FEATURES\n        if len(top_features) >= 4:\n            # Select 4 features for multi-dimensional operations\n            selected = top_features[:4]\n            if all(f in df.columns for f in selected):\n                # Normalize features\n                normalized = []\n                for feat in selected:\n                    x = df[feat].values\n                    normalized.append(self.safe_normalize(x))\n                \n                # GP Feature: Euclidean distance from origin\n                distance = np.sqrt(sum(x**2 for x in normalized))\n                gp_features.append(distance)\n                feature_names.append('gp_euclidean_distance')\n                \n                # GP Feature: RBF kernel\n                rbf = np.exp(-distance**2 / 2)\n                gp_features.append(rbf)\n                feature_names.append('gp_rbf_kernel')\n                \n                # GP Feature: Polynomial kernel\n                poly_kernel = (1 + sum(x for x in normalized))**2\n                gp_features.append(poly_kernel)\n                feature_names.append('gp_poly_kernel')\n        \n        # 7. RATIO AND PROPORTION GP FEATURES\n        for i in range(0, min(6, len(top_features)), 2):\n            if i + 1 < len(top_features):\n                feat1, feat2 = top_features[i], top_features[i + 1]\n                if feat1 in df.columns and feat2 in df.columns:\n                    x = df[feat1].values\n                    y = df[feat2].values\n                    \n                    # GP Feature: Log ratio\n                    log_ratio = np.log(np.abs(x) + 1) - np.log(np.abs(y) + 1)\n                    gp_features.append(log_ratio)\n                    feature_names.append(f'gp_logratio_{feat1}_{feat2}')\n                    \n                    # GP Feature: Harmonic mean\n                    harmonic = 2 * x * y / (x + y + 1e-8)\n                    gp_features.append(harmonic)\n                    feature_names.append(f'gp_harmonic_{feat1}_{feat2}')\n        \n        # Convert to array\n        if gp_features:\n            gp_array = np.column_stack(gp_features)\n            print(f\"  Generated {len(feature_names)} GP features\")\n            return gp_array, feature_names\n        else:\n            return np.zeros((len(df), 0)), []\n\n# ==============================================================================\n# PART 2: ENHANCED FEATURE ENGINEER WITH GP\n# ==============================================================================\n\nclass EnhancedFeatureEngineer:\n    \"\"\"\n    Combines targeted features with genetic programming features.\n    \"\"\"\n    \n    def __init__(self):\n        self.gp_generator = GeneticProgrammingFeatures()\n        self.feature_scaler = RobustScaler()\n        \n    def select_features_by_fitness(self, features, labels, top_k=50):\n        \"\"\"\n        Select top features based on correlation (fitness).\n        \"\"\"\n        correlations = []\n        for i in range(features.shape[1]):\n            corr = abs(pearsonr(features[:, i], labels)[0])\n            correlations.append(corr)\n        \n        # Get indices of top features\n        top_indices = np.argsort(correlations)[::-1][:top_k]\n        return top_indices, [correlations[i] for i in top_indices]\n    \n    def create_all_features(self, df, base_features, labels=None):\n        \"\"\"\n        Create both targeted and GP features, with optional selection.\n        \"\"\"\n        print(\"\\nCreating enhanced feature set...\")\n        \n        all_features = []\n        all_names = []\n        \n        # 1. Original targeted features (from your code)\n        targeted_features, targeted_names = self._create_targeted_features(df, base_features)\n        if targeted_features.shape[1] > 0:\n            all_features.append(targeted_features)\n            all_names.extend(targeted_names)\n        \n        # 2. GP-evolved features\n        gp_features, gp_names = self.gp_generator.create_gp_features(df, base_features[:10])\n        if gp_features.shape[1] > 0:\n            all_features.append(gp_features)\n            all_names.extend(gp_names)\n        \n        # Combine all features\n        if all_features:\n            combined = np.hstack(all_features)\n            \n            # Feature selection if labels provided\n            if labels is not None and len(combined[0]) > 50:\n                print(f\"  Selecting top features from {len(all_names)} candidates...\")\n                top_indices, fitness_scores = self.select_features_by_fitness(combined, labels, top_k=30)\n                \n                selected_features = combined[:, top_indices]\n                selected_names = [all_names[i] for i in top_indices]\n                \n                print(f\"  Selected {len(selected_names)} features with fitness > {min(fitness_scores):.4f}\")\n                return selected_features, selected_names\n            else:\n                return combined, all_names\n        else:\n            return np.zeros((len(df), 0)), []\n    \n    def _create_targeted_features(self, df, base_features):\n        \"\"\"Original targeted features from your code.\"\"\"\n        features = []\n        names = []\n        \n        # Complex interactions\n        if all(f in df.columns for f in ['X862', 'X852', 'X345']):\n            feat1 = df['X862'].values\n            feat2 = df['X852'].values\n            feat3 = df['X345'].values\n            \n            interaction = np.tanh(feat1) * np.exp(-np.abs(feat2) / 2) * np.sign(feat3)\n            features.append(interaction)\n            names.append('complex_interaction_862_852_345')\n            \n            ratio_poly = (feat1 ** 2) / (feat2 ** 2 + 1)\n            features.append(ratio_poly)\n            names.append('poly_ratio_862_852')\n        \n        # Market microstructure\n        if all(f in df.columns for f in ['bid_qty', 'ask_qty', 'volume', 'buy_qty', 'sell_qty']):\n            bid = df['bid_qty'].values\n            ask = df['ask_qty'].values\n            vol = df['volume'].values\n            buy = df['buy_qty'].values\n            sell = df['sell_qty'].values\n            \n            order_imbalance = (bid - ask) / (bid + ask + 1e-6)\n            flow_imbalance = (buy - sell) / (buy + sell + 1e-6)\n            kyle_lambda = flow_imbalance * np.sqrt(np.abs(order_imbalance)) / (np.log1p(vol) + 1e-6)\n            features.append(kyle_lambda)\n            names.append('kyle_lambda_complex')\n            \n            total_pressure = bid + ask\n            vol_adj_pressure = np.log1p(total_pressure) * np.exp(-vol / (vol.mean() + 1e-6))\n            features.append(vol_adj_pressure)\n            names.append('vol_adjusted_pressure')\n        \n        if features:\n            return np.column_stack(features), names\n        else:\n            return np.zeros((len(df), 0)), []\n\n# ==============================================================================\n# PART 3: ENHANCED DUAL MODEL PREDICTOR\n# ==============================================================================\n\nclass GeneticProgrammingCryptoPredictor:\n    \"\"\"Enhanced predictor with genetic programming features.\"\"\"\n    \n    def __init__(self):\n        # Paths\n        self.TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n        self.TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n        self.SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n        \n        # Original features\n        self.ORIGINAL_FEATURES = [\n            \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n            \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n            \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\", \"X888\", \"X421\", \"X333\"\n        ]\n        \n        self.LABEL_COLUMN = \"label\"\n        self.N_FOLDS = 3\n        self.RANDOM_STATE = 42\n        \n        # XGBoost parameters (unchanged)\n        self.XGB_PARAMS = {\n            \"tree_method\": \"hist\",\n            \"device\": \"cpu\",\n            \"colsample_bylevel\": 0.4778,\n            \"colsample_bynode\": 0.3628,\n            \"colsample_bytree\": 0.7107,\n            \"gamma\": 1.7095,\n            \"learning_rate\": 0.02213,\n            \"max_depth\": 20,\n            \"max_leaves\": 12,\n            \"min_child_weight\": 16,\n            \"n_estimators\": 1667,\n            \"subsample\": 0.06567,\n            \"reg_alpha\": 39.3524,\n            \"reg_lambda\": 75.4484,\n            \"verbosity\": 0,\n            \"random_state\": self.RANDOM_STATE,\n            \"n_jobs\": -1,\n            \"verbose\": False,\n        }\n        \n        self.MODEL_SLICES = [\n            {\"name\": \"full_data\", \"cutoff\": 0},\n            {\"name\": \"last_75pct\", \"cutoff\": 0},  \n            {\"name\": \"last_50pct\", \"cutoff\": 0}\n        ]\n        \n        self.feature_engineer = EnhancedFeatureEngineer()\n        self.scaler = RobustScaler()\n        \n    def create_time_decay_weights(self, n, decay=0.95):\n        \"\"\"Create time decay weights for samples.\"\"\"\n        positions = np.arange(n)\n        normalized = positions / float(n - 1)\n        weights = decay ** (1.0 - normalized)\n        return weights * n / weights.sum()\n    \n    def load_data(self):\n        \"\"\"Load train, test, and submission data.\"\"\"\n        print(\"\\nLoading data...\")\n        train_df = pd.read_parquet(self.TRAIN_PATH).reset_index(drop=True)\n        test_df = pd.read_parquet(self.TEST_PATH).reset_index(drop=True)\n        submission_df = pd.read_csv(self.SUBMISSION_PATH)\n        \n        print(f\"Loaded train: {train_df.shape}, test: {test_df.shape}\")\n        return train_df, test_df, submission_df\n    \n    def train_and_predict(self, train_df, test_df, features, model_name=\"Model\"):\n        \"\"\"Train XGBoost model with cross-validation and make predictions.\"\"\"\n        print(f\"\\nTraining {model_name}...\")\n        \n        n_samples = len(train_df)\n        \n        # Set slice cutoffs\n        self.MODEL_SLICES[1][\"cutoff\"] = int(0.25 * n_samples)\n        self.MODEL_SLICES[2][\"cutoff\"] = int(0.50 * n_samples)\n        \n        # Prepare storage\n        oof_preds = {sl[\"name\"]: np.zeros(n_samples) for sl in self.MODEL_SLICES}\n        test_preds = {sl[\"name\"]: np.zeros(len(test_df)) for sl in self.MODEL_SLICES}\n        \n        full_weights = self.create_time_decay_weights(n_samples)\n        kf = KFold(n_splits=self.N_FOLDS, shuffle=False)\n        \n        # Cross-validation\n        for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), start=1):\n            print(f\"\\n--- {model_name} Fold {fold}/{self.N_FOLDS} ---\")\n            X_valid = train_df.iloc[valid_idx][features]\n            y_valid = train_df.iloc[valid_idx][self.LABEL_COLUMN]\n            \n            for sl in self.MODEL_SLICES:\n                slice_name = sl[\"name\"]\n                cutoff = sl[\"cutoff\"]\n                subset = train_df.iloc[cutoff:].reset_index(drop=True)\n                rel_idx = train_idx[train_idx >= cutoff] - cutoff\n                \n                print(f\"Training {slice_name}...\")\n                X_train = subset.iloc[rel_idx][features]\n                y_train = subset.iloc[rel_idx][self.LABEL_COLUMN]\n                \n                # Sample weights\n                if cutoff == 0:\n                    sw = full_weights[train_idx]\n                else:\n                    sw_total = self.create_time_decay_weights(len(subset))\n                    sw = sw_total[rel_idx]\n                \n                # Train model\n                model = XGBRegressor(**self.XGB_PARAMS)\n                model.fit(\n                    X_train, y_train, \n                    sample_weight=sw,\n                    eval_set=[(X_valid, y_valid)],\n                    verbose=100\n                )\n                \n                # OOF predictions\n                mask = valid_idx >= cutoff\n                if mask.any():\n                    idxs = valid_idx[mask]\n                    oof_preds[slice_name][idxs] = model.predict(train_df.iloc[idxs][features])\n                if cutoff > 0 and (~mask).any():\n                    oof_preds[slice_name][valid_idx[~mask]] = oof_preds[\"full_data\"][valid_idx[~mask]]\n                \n                # Test predictions\n                test_preds[slice_name] += model.predict(test_df[features])\n        \n        # Average test predictions\n        for slice_name in test_preds:\n            test_preds[slice_name] /= self.N_FOLDS\n        \n        # Compute Pearson scores\n        pearson_scores = {\n            slice_name: pearsonr(train_df[self.LABEL_COLUMN], preds)[0]\n            for slice_name, preds in oof_preds.items()\n        }\n        \n        print(f\"\\n{model_name} Pearson scores by slice:\")\n        for slice_name, score in pearson_scores.items():\n            print(f\"  {slice_name}: {score:.4f}\")\n        \n        # Weighted ensemble based on performance\n        weights = np.array([pearson_scores[sl[\"name\"]] for sl in self.MODEL_SLICES])\n        weights = weights / weights.sum()\n        \n        oof_ensemble = np.sum([w * oof_preds[sl[\"name\"]] for w, sl in zip(weights, self.MODEL_SLICES)], axis=0)\n        test_ensemble = np.sum([w * test_preds[sl[\"name\"]] for w, sl in zip(weights, self.MODEL_SLICES)], axis=0)\n        ensemble_score = pearsonr(train_df[self.LABEL_COLUMN], oof_ensemble)[0]\n        \n        print(f\"\\n{model_name} Weighted Ensemble Pearson score: {ensemble_score:.4f}\")\n        print(f\"Ensemble weights: {weights}\")\n        \n        return test_ensemble, ensemble_score\n    \n    def prepare_gp_enhanced_data(self, train_df, test_df):\n        \"\"\"Prepare data with GP-enhanced features.\"\"\"\n        print(\"\\nPreparing GP-enhanced features...\")\n        \n        # Create enhanced features with selection\n        train_enhanced, feature_names = self.feature_engineer.create_all_features(\n            train_df, self.ORIGINAL_FEATURES, train_df[self.LABEL_COLUMN].values\n        )\n        test_enhanced, _ = self.feature_engineer.create_all_features(\n            test_df, self.ORIGINAL_FEATURES, None\n        )\n        \n        # Combine with original features\n        train_combined = np.hstack([train_df[self.ORIGINAL_FEATURES].values, train_enhanced])\n        test_combined = np.hstack([test_df[self.ORIGINAL_FEATURES].values, test_enhanced])\n        \n        # Scale all features\n        train_scaled = self.scaler.fit_transform(train_combined)\n        test_scaled = self.scaler.transform(test_combined)\n        \n        # Create DataFrames\n        all_features = self.ORIGINAL_FEATURES + feature_names\n        train_final = pd.DataFrame(train_scaled, columns=all_features)\n        train_final['label'] = train_df['label'].values\n        \n        test_final = pd.DataFrame(test_scaled, columns=all_features)\n        \n        return train_final, test_final, all_features\n\n# ==============================================================================\n# PART 4: MAIN EXECUTION\n# ==============================================================================\n\ndef main():\n    \"\"\"Main execution function.\"\"\"\n    \n    # Initialize predictor\n    predictor = GeneticProgrammingCryptoPredictor()\n    \n    # Load data\n    train_df, test_df, submission_df = predictor.load_data()\n    \n    # =========================================================================\n    # MODEL 1: ORIGINAL MODEL\n    # =========================================================================\n    print(\"\\n\" + \"=\" * 80)\n    print(\"MODEL 1: ORIGINAL MODEL\")\n    print(\"=\" * 80)\n    \n    # Prepare original data\n    train_original = train_df[predictor.ORIGINAL_FEATURES + ['label']].copy()\n    test_original = test_df[predictor.ORIGINAL_FEATURES].copy()\n    \n    # Train original model\n    original_predictions, original_score = predictor.train_and_predict(\n        train_original, test_original, predictor.ORIGINAL_FEATURES, \"Original Model\"\n    )\n    \n    # =========================================================================\n    # MODEL 2: GP-ENHANCED MODEL\n    # =========================================================================\n    print(\"\\n\" + \"=\" * 80)\n    print(\"MODEL 2: GENETIC PROGRAMMING ENHANCED MODEL\")\n    print(\"=\" * 80)\n    \n    # Prepare GP-enhanced data\n    train_gp, test_gp, gp_features = predictor.prepare_gp_enhanced_data(\n        train_df, test_df\n    )\n    \n    print(f\"\\nTotal features in GP model: {len(gp_features)}\")\n    print(f\"GP features added: {len(gp_features) - len(predictor.ORIGINAL_FEATURES)}\")\n    \n    # Train GP-enhanced model\n    gp_predictions, gp_score = predictor.train_and_predict(\n        train_gp, test_gp, gp_features, \"GP-Enhanced Model\"\n    )\n    \n    # =========================================================================\n    # ENSEMBLE: WEIGHTED AVERAGE\n    # =========================================================================\n    print(\"\\n\" + \"=\" * 80)\n    print(\"FINAL ENSEMBLE\")\n    print(\"=\" * 80)\n    \n    # Weight based on OOF scores\n    total_score = original_score + gp_score\n    w_original = original_score / total_score\n    w_gp = gp_score / total_score\n    \n    ensemble_predictions = w_original * original_predictions + w_gp * gp_predictions\n    \n    print(f\"\\nEnsemble weights:\")\n    print(f\"  Original model: {w_original:.3f}\")\n    print(f\"  GP-enhanced model: {w_gp:.3f}\")\n    \n    # Save predictions\n    submission_df[\"prediction\"] = ensemble_predictions\n    submission_df.to_csv(\"submission.csv\", index=False)\n    print(\"\\nSaved submission.csv\")\n    \n    # =========================================================================\n    # RESULTS SUMMARY\n    # =========================================================================\n    print(\"\\n\" + \"=\" * 80)\n    print(\"RESULTS SUMMARY\")\n    print(\"=\" * 80)\n    print(f\"Original Model:\")\n    print(f\"  Features: {len(predictor.ORIGINAL_FEATURES)}\")\n    print(f\"  Score: {original_score:.4f}\")\n    print(f\"\\nGP-Enhanced Model:\")\n    print(f\"  Features: {len(gp_features)}\")\n    print(f\"  Score: {gp_score:.4f}\")\n    print(f\"\\nExpected Ensemble Performance:\")\n    print(f\"  Weighted average: {w_original * original_score + w_gp * gp_score:.4f}\")\n    print(f\"  Potential improvement: +{max(0, gp_score - original_score):.4f}\")\n    \n    # Feature analysis\n    print(\"\\n\" + \"=\" * 80)\n    print(\"GP FEATURE DISCOVERY INSIGHTS\")\n    print(\"=\" * 80)\n    \n    # Quick feature importance check\n    sample_model = XGBRegressor(n_estimators=100, max_depth=6, random_state=42)\n    sample_size = min(50000, len(train_gp))\n    sample_idx = np.random.choice(len(train_gp), sample_size, replace=False)\n    \n    sample_model.fit(\n        train_gp.iloc[sample_idx][gp_features],\n        train_gp.iloc[sample_idx]['label']\n    )\n    \n    importance_df = pd.DataFrame({\n        'feature': gp_features,\n        'importance': sample_model.feature_importances_\n    }).sort_values('importance', ascending=False)\n    \n    # Show top GP features\n    print(\"\\nTop 10 GP-discovered features:\")\n    gp_only = [f for f in gp_features if f not in predictor.ORIGINAL_FEATURES]\n    for i, feat in enumerate(gp_only[:10]):\n        imp = importance_df[importance_df['feature'] == feat]['importance'].values[0]\n        rank = list(importance_df['feature']).index(feat) + 1\n        print(f\"  {i+1}. {feat}: Rank {rank}/{len(gp_features)}, Importance: {imp:.4f}\")\n    \n    print(\"\\n\" + \"=\" * 80)\n    print(\"✨ Genetic Programming feature engineering complete!\")\n    print(\"=\" * 80)\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}