{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport shap\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr, spearmanr\nfrom sklearn.ensemble import RandomForestRegressor\nimport warnings\nwarnings.filterwarnings('ignore')\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n# ==================== UTILITY FUNCTIONS ====================\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print(f'Reducing memory usage for: {dataset}')\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        \n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print(f'--- Memory usage before: {initial_mem_usage:.2f} MB')\n    print(f'--- Memory usage after: {final_mem_usage:.2f} MB')\n    print(f'--- Decreased memory usage by {100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage:.1f}%\\n')\n    \n    return dataframe\n\ndef create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"Create exponentially decaying weights for time-based importance.\"\"\"\n    positions = np.arange(n_samples)\n    normalized_positions = positions / (n_samples - 1)\n    weights = decay_factor ** (1 - normalized_positions)\n    weights = weights * n_samples / weights.sum()\n    return weights\n\ndef remove_highly_correlated_features(df, features, threshold=0.95):\n    \"\"\"Remove features with high correlation to reduce redundancy.\"\"\"\n    print(f\"\\nRemoving highly correlated features (threshold={threshold})...\")\n    corr_matrix = df[features].corr().abs()\n    upper_triangle = np.triu(np.ones_like(corr_matrix, dtype=bool), k=1)\n    \n    to_drop = set()\n    for i in range(len(corr_matrix.columns)):\n        for j in range(i+1, len(corr_matrix.columns)):\n            if corr_matrix.iloc[i, j] > threshold:\n                # Drop the feature with lower correlation to target if available\n                if 'label' in df.columns:\n                    corr_i = abs(df[corr_matrix.columns[i]].corr(df['label']))\n                    corr_j = abs(df[corr_matrix.columns[j]].corr(df['label']))\n                    to_drop.add(corr_matrix.columns[i] if corr_i < corr_j else corr_matrix.columns[j])\n                else:\n                    to_drop.add(corr_matrix.columns[j])\n    \n    print(f\"Removing {len(to_drop)} highly correlated features\")\n    return [f for f in features if f not in to_drop]\n\n# ==================== ENHANCED FEATURE ENGINEERING ====================\n\ndef create_market_microstructure_features(df):\n    \"\"\"Create domain-specific features for crypto market microstructure.\"\"\"\n    print(\"\\nCreating market microstructure features...\")\n    new_features = {}\n    \n    # Order book features\n    if all(col in df.columns for col in ['bid_qty', 'ask_qty']):\n        new_features['order_book_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-8)\n        new_features['order_book_spread'] = np.log1p(df['ask_qty']) - np.log1p(df['bid_qty'])\n        new_features['order_book_depth'] = df['bid_qty'] + df['ask_qty']\n        new_features['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n    \n    # Trade flow features\n    if all(col in df.columns for col in ['buy_qty', 'sell_qty']):\n        new_features['trade_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n        new_features['net_trade_flow'] = df['buy_qty'] - df['sell_qty']\n        new_features['trade_intensity'] = df['buy_qty'] + df['sell_qty']\n        new_features['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    \n    # Volume features\n    if 'volume' in df.columns:\n        new_features['log_volume'] = np.log1p(df['volume'])\n        if 'trade_intensity' in new_features:\n            new_features['avg_trade_size'] = df['volume'] / (new_features['trade_intensity'] + 1e-8)\n    \n    # Combined liquidity indicators\n    if all(f in new_features for f in ['order_book_depth', 'trade_intensity']):\n        new_features['liquidity_ratio'] = new_features['trade_intensity'] / (new_features['order_book_depth'] + 1e-8)\n    \n    return pd.DataFrame(new_features, index=df.index)\n\ndef create_pca_features(train_df, test_df, features, n_components=30):\n    \"\"\"Create PCA features based on EDA insights.\"\"\"\n    print(f\"\\nCreating PCA features (n_components={n_components})...\")\n    \n    # Standardize features\n    scaler = StandardScaler()\n    train_scaled = scaler.fit_transform(train_df[features])\n    test_scaled = scaler.transform(test_df[features])\n    \n    # Apply PCA\n    pca = PCA(n_components=n_components, random_state=42)\n    train_pca = pca.fit_transform(train_scaled)\n    test_pca = pca.transform(test_scaled)\n    \n    # Create PCA feature names\n    pca_features = [f'PCA_{i+1}' for i in range(n_components)]\n    \n    # Convert to DataFrame\n    train_pca_df = pd.DataFrame(train_pca, columns=pca_features, index=train_df.index)\n    test_pca_df = pd.DataFrame(test_pca, columns=pca_features, index=test_df.index)\n    \n    print(f\"PCA explained variance ratio: {pca.explained_variance_ratio_.sum():.4f}\")\n    \n    return train_pca_df, test_pca_df\n\ndef create_cluster_features(train_df, test_df, features, n_clusters=5):\n    \"\"\"Create cluster-based features from EDA insights.\"\"\"\n    print(f\"\\nCreating cluster features (n_clusters={n_clusters})...\")\n    \n    # Standardize features\n    scaler = StandardScaler()\n    train_scaled = scaler.fit_transform(train_df[features])\n    test_scaled = scaler.transform(test_df[features])\n    \n    # Apply K-means clustering\n    kmeans = KMeans(n_clusters=n_clusters, random_state=42, n_init=10)\n    train_clusters = kmeans.fit_predict(train_scaled)\n    test_clusters = kmeans.predict(test_scaled)\n    \n    # Calculate distances to each cluster center\n    train_distances = kmeans.transform(train_scaled)\n    test_distances = kmeans.transform(test_scaled)\n    \n    # Create cluster features\n    cluster_features = {}\n    cluster_features['cluster'] = train_clusters\n    \n    for i in range(n_clusters):\n        cluster_features[f'dist_to_cluster_{i}'] = train_distances[:, i]\n    \n    train_cluster_df = pd.DataFrame(cluster_features, index=train_df.index)\n    \n    # Test features\n    test_cluster_features = {}\n    test_cluster_features['cluster'] = test_clusters\n    \n    for i in range(n_clusters):\n        test_cluster_features[f'dist_to_cluster_{i}'] = test_distances[:, i]\n    \n    test_cluster_df = pd.DataFrame(test_cluster_features, index=test_df.index)\n    \n    return train_cluster_df, test_cluster_df\n\ndef create_interaction_features(df, top_features, max_interactions=50):\n    \"\"\"Create interaction features for top performing features.\"\"\"\n    print(f\"\\nCreating interaction features for top {len(top_features)} features...\")\n    new_features = {}\n    \n    # Create interactions only for top features to manage memory\n    interaction_count = 0\n    for i, feat1 in enumerate(top_features):\n        if interaction_count >= max_interactions:\n            break\n        for j, feat2 in enumerate(top_features[i+1:], i+1):\n            if interaction_count >= max_interactions:\n                break\n            \n            # Multiplicative interaction\n            new_features[f'{feat1}_x_{feat2}'] = df[feat1] * df[feat2]\n            \n            # Ratio interaction (with safety for division)\n            denominator = df[feat2].replace(0, np.nan)\n            new_features[f'{feat1}_div_{feat2}'] = df[feat1] / denominator\n            new_features[f'{feat1}_div_{feat2}'].fillna(0, inplace=True)\n            \n            interaction_count += 2\n    \n    print(f\"Created {len(new_features)} interaction features\")\n    return pd.DataFrame(new_features, index=df.index)\n\n# ==================== FEATURE SELECTION ====================\n\ndef select_features_ensemble(train_df, test_df, target='label'):\n    \"\"\"Enhanced feature selection based on EDA insights.\"\"\"\n    print(\"\\n\" + \"=\"*60)\n    print(\"ENHANCED FEATURE SELECTION\")\n    print(\"=\"*60)\n    \n    # Define feature groups based on EDA\n    # Top correlated features from EDA\n    top_correlated = ['X21', 'X20', 'X28', 'X863', 'X29', 'X19', 'X27', 'X22', \n                      'X858', 'X219', 'X860', 'X531', 'X287', 'X289', 'X291']\n    \n    # Features selected by all methods in EDA\n    consensus_features = ['X175', 'X179', 'X137', 'X197', 'X22', 'X40', 'X181', \n                         'X28', 'X169', 'X198', 'X173']\n    \n    # Original selected features from simpler model\n    original_selected = [\"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \n                        \"X603\", \"X860\", \"X674\", \"X415\", \"X345\", \"X137\", \"X855\", \n                        \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\"]\n    \n    # Market features\n    market_features = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    \n    # Combine all important features\n    all_important = list(set(top_correlated + consensus_features + original_selected))\n    \n    # Get additional X features not in our important list\n    all_x_features = [col for col in train_df.columns if col.startswith('X')]\n    remaining_x_features = [f for f in all_x_features if f not in all_important]\n    \n    # Calculate correlations for remaining features\n    print(\"Evaluating remaining features...\")\n    y = train_df[target]\n    feature_scores = []\n    \n    for feat in remaining_x_features[:200]:  # Evaluate top 200 remaining\n        try:\n            corr = abs(pearsonr(train_df[feat], y)[0])\n            feature_scores.append((feat, corr))\n        except:\n            continue\n    \n    # Sort and select top additional features\n    feature_scores.sort(key=lambda x: x[1], reverse=True)\n    additional_features = [feat for feat, _ in feature_scores[:15]]\n    \n    # Final feature list\n    selected_features = all_important + additional_features + market_features\n    selected_features = list(set(selected_features))  # Remove duplicates\n    \n    # Remove highly correlated features\n    selected_features = remove_highly_correlated_features(\n        train_df, selected_features, threshold=0.95\n    )\n    \n    print(f\"\\nSelected {len(selected_features)} features\")\n    print(f\"  - Top correlated: {len([f for f in selected_features if f in top_correlated])}\")\n    print(f\"  - Consensus features: {len([f for f in selected_features if f in consensus_features])}\")\n    print(f\"  - Market features: {len([f for f in selected_features if f in market_features])}\")\n    \n    return selected_features\n\n# ==================== MODEL CONFIGURATIONS ====================\n\ndef get_model_params():\n    \"\"\"Define optimized parameters for each boosting algorithm.\"\"\"\n    \n    xgb_params = {\n        \"tree_method\": \"gpu_hist\" if \"gpu\" in str(pd.__version__) else \"hist\",\n        \"n_estimators\": 1500,\n        \"learning_rate\": 0.02,\n        \"max_depth\": 15,\n        \"max_leaves\": 20,\n        \"min_child_weight\": 20,\n        \"subsample\": 0.8,\n        \"colsample_bytree\": 0.7,\n        \"colsample_bylevel\": 0.6,\n        \"colsample_bynode\": 0.5,\n        \"gamma\": 2.0,\n        \"reg_alpha\": 40,\n        \"reg_lambda\": 80,\n        \"random_state\": 42,\n        \"n_jobs\": -1,\n        \"verbosity\": 0\n    }\n    \n    catboost_params = {\n        \"iterations\": 1500,\n        \"learning_rate\": 0.02,\n        \"depth\": 10,\n        \"l2_leaf_reg\": 50,\n        \"min_data_in_leaf\": 30,\n        \"random_strength\": 2.0,\n        \"bagging_temperature\": 0.8,\n        \"border_count\": 128,\n        \"subsample\": 0.8,\n        \"sampling_frequency\": \"PerTree\",\n        \"random_state\": 42,\n        \"verbose\": False,\n        \"thread_count\": -1,\n        \"task_type\": \"GPU\" if \"gpu\" in str(pd.__version__) else \"CPU\"\n    }\n    \n    lgbm_params = {\n        \"n_estimators\": 1500,\n        \"learning_rate\": 0.02,\n        \"num_leaves\": 31,\n        \"max_depth\": 12,\n        \"min_child_samples\": 30,\n        \"subsample\": 0.8,\n        \"subsample_freq\": 1,\n        \"colsample_bytree\": 0.7,\n        \"reg_alpha\": 40,\n        \"reg_lambda\": 80,\n        \"random_state\": 42,\n        \"n_jobs\": -1,\n        \"verbosity\": -1,\n        \"metric\": \"rmse\",\n        \"device\": \"gpu\" if \"gpu\" in str(pd.__version__) else \"cpu\"\n    }\n    \n    return xgb_params, catboost_params, lgbm_params\n\n# ==================== TRAINING PIPELINE ====================\n\ndef train_model_ensemble(train_df, test_df, features, target='label'):\n    \"\"\"Train ensemble of XGBoost, CatBoost, and LightGBM models.\"\"\"\n    \n    print(\"\\n\" + \"=\"*70)\n    print(\"TRAINING MULTI-ALGORITHM ENSEMBLE\")\n    print(\"=\"*70)\n    \n    # Get model parameters\n    xgb_params, catboost_params, lgbm_params = get_model_params()\n    \n    # Define cross-validation\n    FOLDS = 5\n    kf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n    \n    # Model configurations for time-based training\n    time_configs = [\n        {\"name\": \"Full Data\", \"percent\": 1.00},\n        {\"name\": \"Recent 80%\", \"percent\": 0.80},\n        {\"name\": \"Recent 60%\", \"percent\": 0.60},\n        {\"name\": \"Recent 40%\", \"percent\": 0.40}\n    ]\n    \n    # Initialize predictions storage\n    model_types = ['xgb', 'catboost', 'lgbm']\n    oof_predictions = {model: {config['name']: np.zeros(len(train_df)) \n                              for config in time_configs} \n                      for model in model_types}\n    test_predictions = {model: {config['name']: np.zeros(len(test_df)) \n                               for config in time_configs} \n                       for model in model_types}\n    \n    # Train models\n    for fold_num, (train_idx, valid_idx) in enumerate(kf.split(train_df)):\n        print(f\"\\n{'='*50}\")\n        print(f\"FOLD {fold_num + 1}\")\n        print(f\"{'='*50}\")\n        \n        X_valid = train_df.iloc[valid_idx][features]\n        y_valid = train_df.iloc[valid_idx][target]\n        X_test = test_df[features]\n        \n        for config in time_configs:\n            print(f\"\\n--- {config['name']} ---\")\n            \n            # Calculate data subset\n            if config[\"percent\"] == 1.00:\n                X_train = train_df.iloc[train_idx][features]\n                y_train = train_df.iloc[train_idx][target]\n                sample_weights = create_time_weights(len(X_train), decay_factor=0.95)\n            else:\n                cutoff_idx = int(len(train_df) * (1 - config[\"percent\"]))\n                train_idx_recent = train_idx[train_idx >= cutoff_idx]\n                train_idx_adjusted = train_idx_recent - cutoff_idx\n                train_recent = train_df.iloc[cutoff_idx:].reset_index(drop=True)\n                \n                X_train = train_recent.iloc[train_idx_adjusted][features]\n                y_train = train_recent.iloc[train_idx_adjusted][target]\n                sample_weights = create_time_weights(len(X_train), decay_factor=0.95)\n            \n            # Train XGBoost\n            print(\"Training XGBoost...\")\n            xgb_model = XGBRegressor(**xgb_params)\n            xgb_model.fit(\n                X_train, y_train,\n                sample_weight=sample_weights,\n                eval_set=[(X_valid, y_valid)],\n                early_stopping_rounds=50,\n                verbose=False\n            )\n            oof_predictions['xgb'][config['name']][valid_idx] = xgb_model.predict(X_valid)\n            test_predictions['xgb'][config['name']] += xgb_model.predict(X_test) / FOLDS\n            \n            # Train CatBoost\n            print(\"Training CatBoost...\")\n            catboost_model = CatBoostRegressor(**catboost_params)\n            catboost_model.fit(\n                X_train, y_train,\n                sample_weight=sample_weights,\n                eval_set=(X_valid, y_valid),\n                early_stopping_rounds=50,\n                verbose=False\n            )\n            oof_predictions['catboost'][config['name']][valid_idx] = catboost_model.predict(X_valid)\n            test_predictions['catboost'][config['name']] += catboost_model.predict(X_test) / FOLDS\n            \n            # Train LightGBM\n            print(\"Training LightGBM...\")\n            lgbm_model = LGBMRegressor(**lgbm_params)\n            lgbm_model.fit(\n                X_train, y_train,\n                sample_weight=sample_weights,\n                eval_set=[(X_valid, y_valid)],\n                callbacks=[],\n                eval_metric='rmse'\n            )\n            oof_predictions['lgbm'][config['name']][valid_idx] = lgbm_model.predict(X_valid)\n            test_predictions['lgbm'][config['name']] += lgbm_model.predict(X_test) / FOLDS\n    \n    return oof_predictions, test_predictions\n\n# ==================== MAIN EXECUTION ====================\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"ENHANCED CRYPTO PRICE PREDICTION WITH MULTI-ALGORITHM ENSEMBLE\")\nprint(\"=\"*80)\n\n# Load data\nprint(\"\\nLoading data...\")\ntrain = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\nsample = pd.read_csv(CFG.sample_sub_path)\n\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Test shape: {test.shape}\")\n\n# Feature selection based on EDA insights\nselected_features = select_features_ensemble(train, test)\n\n# Create market microstructure features\ntrain_market = create_market_microstructure_features(train)\ntest_market = create_market_microstructure_features(test)\n\n# Create PCA features (using 30 components based on EDA showing 23 for 90% variance)\ntrain_pca, test_pca = create_pca_features(train, test, selected_features, n_components=30)\n\n# Create cluster features (5 clusters based on EDA)\ntrain_clusters, test_clusters = create_cluster_features(train, test, selected_features, n_clusters=5)\n\n# Create interaction features for top features\ntop_features_for_interaction = selected_features[:15]  # Top 15 features\ntrain_interactions = create_interaction_features(train, top_features_for_interaction)\ntest_interactions = create_interaction_features(test, top_features_for_interaction)\n\n# Combine all features\nprint(\"\\nCombining all features...\")\ntrain_enhanced = pd.concat([\n    train[selected_features + ['label']], \n    train_market, \n    train_pca, \n    train_clusters,\n    train_interactions\n], axis=1)\n\ntest_enhanced = pd.concat([\n    test[selected_features], \n    test_market, \n    test_pca, \n    test_clusters,\n    test_interactions\n], axis=1)\n\n# Get final feature list (excluding label)\nfinal_features = [col for col in train_enhanced.columns if col != 'label']\nprint(f\"\\nFinal feature count: {len(final_features)}\")\n\n# Reduce memory usage\ntrain_enhanced = reduce_mem_usage(train_enhanced, \"train_enhanced\")\ntest_enhanced = reduce_mem_usage(test_enhanced, \"test_enhanced\")\n\n# Train ensemble models\noof_predictions, test_predictions = train_model_ensemble(\n    train_enhanced, test_enhanced, final_features\n)\n\n# Calculate individual model performances\nprint(\"\\n\" + \"=\"*60)\nprint(\"MODEL PERFORMANCE EVALUATION\")\nprint(\"=\"*60)\n\ny_true = train_enhanced['label']\nperformance_scores = {}\n\nfor model_type in ['xgb', 'catboost', 'lgbm']:\n    print(f\"\\n{model_type.upper()} Performance:\")\n    performance_scores[model_type] = {}\n    \n    for config_name in oof_predictions[model_type]:\n        score = pearsonr(y_true, oof_predictions[model_type][config_name])[0]\n        performance_scores[model_type][config_name] = score\n        print(f\"  {config_name}: {score:.4f}\")\n\n# Create final ensemble\nprint(\"\\n\" + \"=\"*60)\nprint(\"CREATING FINAL ENSEMBLE\")\nprint(\"=\"*60)\n\n# Flatten all predictions for ensemble\nall_oof_preds = []\nall_test_preds = []\nall_scores = []\n\nfor model_type in ['xgb', 'catboost', 'lgbm']:\n    for config_name in oof_predictions[model_type]:\n        all_oof_preds.append(oof_predictions[model_type][config_name])\n        all_test_preds.append(test_predictions[model_type][config_name])\n        all_scores.append(performance_scores[model_type][config_name])\n\n# Calculate weights based on performance\nall_scores = np.array(all_scores)\nweights = all_scores / all_scores.sum()\n\n# Create weighted ensemble\nfinal_oof = np.zeros(len(train_enhanced))\nfinal_test = np.zeros(len(test_enhanced))\n\nfor i, (oof_pred, test_pred, weight) in enumerate(zip(all_oof_preds, all_test_preds, weights)):\n    final_oof += weight * oof_pred\n    final_test += weight * test_pred\n\n# Calculate final ensemble score\nfinal_score = pearsonr(y_true, final_oof)[0]\nprint(f\"\\nFinal Weighted Ensemble Score: {final_score:.4f}\")\n\n# Also try simple average ensemble\nsimple_oof = np.mean(all_oof_preds, axis=0)\nsimple_test = np.mean(all_test_preds, axis=0)\nsimple_score = pearsonr(y_true, simple_oof)[0]\nprint(f\"Simple Average Ensemble Score: {simple_score:.4f}\")\n\n# Use the better ensemble\nif final_score > simple_score:\n    final_predictions = final_test\n    print(\"\\nUsing weighted ensemble for final predictions\")\nelse:\n    final_predictions = simple_test\n    print(\"\\nUsing simple average ensemble for final predictions\")\n\n# Feature importance analysis using best single model\nprint(\"\\nGenerating feature importance analysis...\")\n# Train a single XGBoost model on full data for SHAP\nxgb_params, _, _ = get_model_params()\nxgb_full = XGBRegressor(**xgb_params)\nxgb_full.fit(train_enhanced[final_features], y_true, verbose=False)\n\n# Get feature importances\nfeature_importance = pd.DataFrame({\n    'feature': final_features,\n    'importance': xgb_full.feature_importances_\n}).sort_values('importance', ascending=False)\n\nprint(\"\\nTop 20 most important features:\")\nprint(feature_importance.head(20))\n\n# Save predictions\nsample[\"prediction\"] = final_predictions\nsample.to_csv(\"submission.csv\", index=False)\nprint(\"\\nPredictions saved to submission.csv\")\nprint(sample.head())\n\n# Save detailed results\nresults_summary = pd.DataFrame({\n    'Model_Config': [f\"{model}_{config}\" for model in ['xgb', 'catboost', 'lgbm'] \n                     for config in performance_scores[model].keys()],\n    'Pearson_Score': [performance_scores[model][config] \n                      for model in ['xgb', 'catboost', 'lgbm'] \n                      for config in performance_scores[model].keys()],\n    'Weight_in_Ensemble': [w for w in weights]\n})\nresults_summary = results_summary.sort_values('Pearson_Score', ascending=False)\nresults_summary.to_csv(\"ensemble_results_enhanced.csv\", index=False)\nprint(\"\\nDetailed results saved to ensemble_results_enhanced.csv\")\n\n# Save feature list\npd.DataFrame({'feature': final_features}).to_csv(\"final_features_enhanced.csv\", index=False)\nprint(f\"\\nFinal feature list saved. Total features used: {len(final_features)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}