{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport shap\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr, spearmanr\nfrom sklearn.ensemble import RandomForestRegressor\nimport warnings\nwarnings.filterwarnings('ignore')\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\ndef create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.95 = 5% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    return weights\n\ndef adjust_xgb_params_for_features(base_params, n_features):\n    \"\"\"\n    Dynamically adjust XGBoost parameters based on number of features\n    to prevent overfitting when adding more features\n    \"\"\"\n    params = base_params.copy()\n    \n    # As we add more features, we need more regularization\n    if n_features > 25:\n        # Increase regularization\n        regularization_factor = 1 + (n_features - 25) * 0.1\n        params['reg_alpha'] = base_params['reg_alpha'] * regularization_factor\n        params['reg_lambda'] = base_params['reg_lambda'] * regularization_factor\n        \n        # Slightly reduce tree complexity\n        params['max_depth'] = max(12, base_params['max_depth'] - (n_features - 25) // 10)\n        params['max_leaves'] = max(8, base_params['max_leaves'] - (n_features - 25) // 15)\n        \n        # Increase minimum child weight\n        params['min_child_weight'] = base_params['min_child_weight'] + (n_features - 25) // 5\n        \n        # Reduce column sampling\n        params['colsample_bytree'] = max(0.5, base_params['colsample_bytree'] - (n_features - 25) * 0.01)\n        \n        print(f\"\\nAdjusted parameters for {n_features} features:\")\n        print(f\"  reg_alpha: {base_params['reg_alpha']:.1f} -> {params['reg_alpha']:.1f}\")\n        print(f\"  reg_lambda: {base_params['reg_lambda']:.1f} -> {params['reg_lambda']:.1f}\")\n        print(f\"  max_depth: {base_params['max_depth']} -> {params['max_depth']}\")\n        print(f\"  colsample_bytree: {base_params['colsample_bytree']:.3f} -> {params['colsample_bytree']:.3f}\")\n    \n    return params\n\ndef robust_feature_selection(train_df, test_df, current_features, target='label', max_new_features=5):\n    \"\"\"\n    Robust feature selection that validates features using cross-validation\n    and checks for redundancy\n    \"\"\"\n    print(\"\\n\" + \"=\"*60)\n    print(\"ROBUST FEATURE SELECTION WITH OVERFITTING PREVENTION\")\n    print(\"=\"*60)\n    \n    # Get all X features not in current selection\n    all_x_features = [col for col in train_df.columns if col.startswith('X')]\n    available_features = [f for f in all_x_features if f not in current_features]\n    \n    print(f\"Current features: {len(current_features)}\")\n    print(f\"Available X features to evaluate: {len(available_features)}\")\n    \n    # Prepare data\n    X_current = train_df[current_features]\n    y = train_df[target]\n    \n    # Calculate baseline performance with current features\n    print(\"\\nCalculating baseline performance...\")\n    baseline_model = XGBRegressor(n_estimators=100, max_depth=5, random_state=42)\n    baseline_scores = cross_val_score(baseline_model, X_current, y, cv=5, \n                                    scoring='neg_mean_squared_error')\n    baseline_score = -np.mean(baseline_scores)\n    print(f\"Baseline MSE with {len(current_features)} features: {baseline_score:.6f}\")\n    \n    # Evaluate each candidate feature\n    feature_evaluations = []\n    \n    print(\"\\nEvaluating candidate features...\")\n    for i, feat in enumerate(available_features[:200]):  # Limit to 200 for speed\n        if i % 20 == 0:\n            print(f\"Progress: {i}/{min(200, len(available_features))} features evaluated\")\n        \n        try:\n            # Clean the feature\n            train_feat = train_df[feat].replace([np.inf, -np.inf], np.nan)\n            test_feat = test_df[feat].replace([np.inf, -np.inf], np.nan)\n            \n            # Skip if too many missing values\n            if train_feat.isna().sum() > len(train_feat) * 0.3:\n                continue\n            \n            train_feat = train_feat.fillna(train_feat.median())\n            test_feat = test_feat.fillna(test_feat.median())\n            \n            # Skip if constant\n            if train_feat.std() < 1e-8:\n                continue\n            \n            # 1. Check redundancy with existing features\n            max_corr_existing = 0\n            for existing_feat in current_features[:20]:  # Check top 20 features\n                if existing_feat in train_df.columns:\n                    corr = abs(pearsonr(train_feat, train_df[existing_feat])[0])\n                    max_corr_existing = max(max_corr_existing, corr)\n            \n            # Skip if too correlated with existing features\n            if max_corr_existing > 0.95:\n                continue\n            \n            # 2. Calculate robust correlation using different methods\n            pearson_corr = abs(pearsonr(train_feat, y)[0])\n            spearman_corr = abs(spearmanr(train_feat, y)[0])\n            \n            # 3. Cross-validated importance\n            X_with_feat = X_current.copy()\n            X_with_feat[feat] = train_feat\n            \n            # Quick CV to test if feature improves performance\n            cv_model = XGBRegressor(n_estimators=100, max_depth=5, random_state=42)\n            cv_scores = cross_val_score(cv_model, X_with_feat, y, cv=3, \n                                      scoring='neg_mean_squared_error')\n            cv_score = -np.mean(cv_scores)\n            \n            # Calculate improvement\n            improvement = (baseline_score - cv_score) / baseline_score\n            \n            # 4. Stability checks\n            # Check correlation stability across time windows\n            n_windows = 3\n            window_size = len(train_df) // n_windows\n            window_corrs = []\n            \n            for w in range(n_windows):\n                start = w * window_size\n                end = start + window_size if w < n_windows - 1 else len(train_df)\n                window_data = train_df.iloc[start:end]\n                if len(window_data) > 100:\n                    w_corr = abs(pearsonr(window_data[feat], window_data[target])[0])\n                    window_corrs.append(w_corr)\n            \n            corr_stability = 1 - (np.std(window_corrs) / (np.mean(window_corrs) + 1e-8)) if window_corrs else 0\n            \n            # 5. Permutation test (simplified)\n            n_perms = 5\n            perm_corrs = []\n            for _ in range(n_perms):\n                y_perm = np.random.permutation(y)\n                perm_corr = abs(pearsonr(train_feat, y_perm)[0])\n                perm_corrs.append(perm_corr)\n            \n            # Feature should have much higher correlation than random\n            signal_ratio = pearson_corr / (np.mean(perm_corrs) + 1e-8)\n            \n            feature_evaluations.append({\n                'feature': feat,\n                'pearson_corr': pearson_corr,\n                'spearman_corr': spearman_corr,\n                'cv_improvement': improvement,\n                'redundancy': max_corr_existing,\n                'stability': corr_stability,\n                'signal_ratio': signal_ratio,\n                'combined_score': (\n                    0.25 * pearson_corr +\n                    0.15 * spearman_corr +\n                    0.30 * max(0, improvement) +\n                    0.10 * (1 - max_corr_existing) +\n                    0.10 * corr_stability +\n                    0.10 * min(signal_ratio / 5, 1)  # Cap signal ratio contribution\n                )\n            })\n            \n        except Exception as e:\n            continue\n    \n    if not feature_evaluations:\n        print(\"No valid features found!\")\n        return []\n    \n    # Sort by combined score\n    eval_df = pd.DataFrame(feature_evaluations).sort_values('combined_score', ascending=False)\n    \n    # Select features that actually improve performance\n    selected_new_features = []\n    for _, row in eval_df.iterrows():\n        if row['cv_improvement'] > 0.001 and row['signal_ratio'] > 2.0 and len(selected_new_features) < max_new_features:\n            selected_new_features.append(row['feature'])\n    \n    print(f\"\\nSelected {len(selected_new_features)} new features that improve performance\")\n    \n    if selected_new_features:\n        print(\"\\nTop selected features:\")\n        for feat in selected_new_features[:5]:\n            feat_info = eval_df[eval_df['feature'] == feat].iloc[0]\n            print(f\"\\n{feat}:\")\n            print(f\"  Correlation: {feat_info['pearson_corr']:.4f}\")\n            print(f\"  CV Improvement: {feat_info['cv_improvement']*100:.2f}%\")\n            print(f\"  Redundancy: {feat_info['redundancy']:.4f}\")\n            print(f\"  Signal Ratio: {feat_info['signal_ratio']:.2f}\")\n    \n    # Save detailed analysis\n    eval_df.to_csv(\"robust_feature_evaluation.csv\", index=False)\n    \n    return selected_new_features\n\n# ==================== MAIN EXECUTION ====================\n\n# Load data\ntrain = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\nsample = pd.read_csv(CFG.sample_sub_path)\n\n# Original selected features\noriginal_features = [\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n]\n\n# Robust feature selection\nnew_features = robust_feature_selection(train, test, original_features, max_new_features=3)\n\n# Decide whether to add features based on validation\nif new_features:\n    print(f\"\\nAdding {len(new_features)} validated features\")\n    selected_features = original_features + new_features\nelse:\n    print(\"\\nNo features passed validation. Using original feature set.\")\n    selected_features = original_features\n\nprint(f\"\\nFinal feature count: {len(selected_features)}\")\n\n# Select features and reduce memory\ntrain = train[selected_features + [\"label\"]]\ntest = test[selected_features]\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nprint(\"Train=\", train.shape)\nprint(\"Test=\", test.shape)\nprint(\"Sample=\", sample.shape)\n\nRMV = [\"label\"]\nFEATURES = [c for c in train.columns if c not in RMV]\nprint(f\"There are {len(FEATURES)} FEATURES\")\n\n# Define cross-validation\nFOLDS = 5\nkf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n\n# Base XGBoost parameters\nbase_xgb_params = {\n    \"tree_method\": \"gpu_hist\",\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}\n\n# Adjust parameters based on feature count\nxgb_params = adjust_xgb_params_for_features(base_xgb_params, len(FEATURES))\n\n# Define model configurations\nmodel_configs = [\n    {\"name\": \"Model 1 (100% Full Data)\", \"percent\": 1.00},\n    {\"name\": \"Model 2 (90% Recent)\", \"percent\": 0.90},\n    {\"name\": \"Model 3 (80% Recent)\", \"percent\": 0.80},\n    {\"name\": \"Model 4 (70% Recent)\", \"percent\": 0.70},\n    {\"name\": \"Model 5 (60% Recent)\", \"percent\": 0.60},\n    {\"name\": \"Model 6 (50% Recent)\", \"percent\": 0.50},\n    {\"name\": \"Model 7 (40% Recent)\", \"percent\": 0.40}\n]\n\n# Initialize predictions for all models\nn_models = len(model_configs)\noof_preds_all = [np.zeros(len(train)) for _ in range(n_models)]\ntest_preds_all = [np.zeros(len(test)) for _ in range(n_models)]\n\n# Generate sample weights for Model 1 (full data)\nsample_weights_full = create_time_weights(len(train), decay_factor=0.95)\nprint(f\"\\nModel 1 - Full data sample weights range: [{sample_weights_full.min():.4f}, {sample_weights_full.max():.4f}]\")\nprint(f\"Model 1 - Full data sample weights mean: {sample_weights_full.mean():.4f}\")\n\n# Calculate cutoffs for each model\ncutoffs = []\nfor config in model_configs:\n    if config[\"percent\"] == 1.00:\n        cutoffs.append(0)\n    else:\n        cutoff_idx = int(len(train) * (1 - config[\"percent\"]))\n        cutoffs.append(cutoff_idx)\n        print(f\"\\n{config['name']} - Using most recent {len(train) - cutoff_idx} samples ({int(config['percent']*100)}% of data)\")\n\n# Cross-validation loop\nfor fold_num, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"\\n\" + \"#\" * 50)\n    print(f\"### Fold {fold_num + 1}\")\n    print(\"#\" * 50)\n    \n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx][\"label\"]\n    X_test = test[FEATURES]\n    \n    # Train each model\n    for model_idx, (config, cutoff) in enumerate(zip(model_configs, cutoffs)):\n        print(f\"\\n--- {config['name']} ---\")\n        \n        if config[\"percent\"] == 1.00:\n            # Model 1: Full data with time weights\n            X_train = train.iloc[train_idx][FEATURES]\n            y_train = train.iloc[train_idx][\"label\"]\n            train_weights = sample_weights_full[train_idx]\n        else:\n            # Other models: Recent data subsets\n            train_idx_recent = train_idx[train_idx >= cutoff]\n            train_idx_recent_adjusted = train_idx_recent - cutoff\n            train_recent = train.iloc[cutoff:].reset_index(drop=True)\n            \n            X_train = train_recent.iloc[train_idx_recent_adjusted][FEATURES]\n            y_train = train_recent.iloc[train_idx_recent_adjusted][\"label\"]\n            \n            sample_weights_recent = create_time_weights(len(train_recent), decay_factor=0.95)\n            train_weights = sample_weights_recent[train_idx_recent_adjusted]\n        \n        # Train the model\n        model = XGBRegressor(**xgb_params)\n        model.fit(\n            X_train, y_train,\n            sample_weight=train_weights,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=25,\n            verbose=200\n        )\n        \n        # Make predictions\n        if config[\"percent\"] == 1.00:\n            oof_preds_all[model_idx][valid_idx] = model.predict(X_valid)\n        else:\n            valid_idx_in_range = valid_idx[valid_idx >= cutoff]\n            if len(valid_idx_in_range) > 0:\n                X_valid_subset = train.iloc[valid_idx_in_range][FEATURES]\n                oof_preds_all[model_idx][valid_idx_in_range] = model.predict(X_valid_subset)\n            \n            valid_idx_out_range = valid_idx[valid_idx < cutoff]\n            if len(valid_idx_out_range) > 0:\n                oof_preds_all[model_idx][valid_idx_out_range] = oof_preds_all[0][valid_idx_out_range]\n        \n        test_preds_all[model_idx] += model.predict(X_test)\n\n# Average test predictions across folds\nfor i in range(n_models):\n    test_preds_all[i] /= FOLDS\n\n# Calculate individual model scores\npearson_scores = []\nfor i, config in enumerate(model_configs):\n    score = pearsonr(train[\"label\"], oof_preds_all[i])[0]\n    pearson_scores.append(score)\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"INDIVIDUAL MODEL PERFORMANCE\")\nprint(\"=\" * 50)\nfor config, score in zip(model_configs, pearson_scores):\n    print(f\"{config['name']} Pearson Correlation: {score:.4f}\")\n\n# Create ensemble predictions\nensemble_oof_preds = np.mean(oof_preds_all, axis=0)\nensemble_test_preds = np.mean(test_preds_all, axis=0)\nensemble_pearson_score = pearsonr(train[\"label\"], ensemble_oof_preds)[0]\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"ENSEMBLE PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Ensemble (Equal Weight) Pearson Correlation: {ensemble_pearson_score:.4f}\")\n\n# Performance-weighted ensemble\ntotal_score = sum(pearson_scores)\nweights = [score / total_score for score in pearson_scores]\n\nweighted_ensemble_oof = np.zeros(len(train))\nweighted_ensemble_test = np.zeros(len(test))\n\nfor i in range(n_models):\n    weighted_ensemble_oof += weights[i] * oof_preds_all[i]\n    weighted_ensemble_test += weights[i] * test_preds_all[i]\n\nweighted_ensemble_score = pearsonr(train[\"label\"], weighted_ensemble_oof)[0]\n\nprint(f\"\\nWeighted Ensemble Performance:\")\nfor config, weight in zip(model_configs, weights):\n    print(f\"  {config['name']} weight: {weight:.3f}\")\nprint(f\"  Weighted Ensemble Pearson Correlation: {weighted_ensemble_score:.4f}\")\n\n# Use the better ensemble for final predictions\nif weighted_ensemble_score > ensemble_pearson_score:\n    final_test_preds = weighted_ensemble_test\n    print(\"\\nUsing weighted ensemble for final predictions\")\nelse:\n    final_test_preds = ensemble_test_preds\n    print(\"\\nUsing simple average ensemble for final predictions\")\n\n# SHAP analysis\nprint(\"\\nGenerating SHAP analysis...\")\nmodel1_for_shap = XGBRegressor(**xgb_params)\nmodel1_for_shap.fit(\n    train[FEATURES], train[\"label\"],\n    sample_weight=sample_weights_full,\n    verbose=0\n)\nexplainer = shap.TreeExplainer(model1_for_shap, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\nshap_values = explainer.shap_values(X_test)\nshap.summary_plot(shap_values, X_test)\n\n# Save predictions\nsample[\"prediction\"] = final_test_preds\nsample.to_csv(\"submission.csv\", index=False)\nprint(\"\\nPredictions saved to submission.csv\")\nprint(sample.head())\n\n# Save detailed results\nresults_data = {\n    'model': [config['name'] for config in model_configs] + ['Simple Ensemble', 'Weighted Ensemble'],\n    'pearson_correlation': pearson_scores + [ensemble_pearson_score, weighted_ensemble_score],\n    'weight_in_final': [weight if weighted_ensemble_score > ensemble_pearson_score else 1/n_models \n                        for weight in weights] + [np.nan, np.nan]\n}\n\nensemble_results = pd.DataFrame(results_data)\nensemble_results.to_csv(\"ensemble_results.csv\", index=False)\nprint(\"\\nEnsemble results saved to ensemble_results.csv\")\nprint(ensemble_results)\n\n# Save feature information\nfeature_info = pd.DataFrame({\n    'feature': FEATURES,\n    'is_original': [feat in original_features for feat in FEATURES]\n})\nfeature_info.to_csv(\"final_features.csv\", index=False)\nprint(f\"\\nFinal feature list saved. Total features: {len(FEATURES)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}