{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Single Variation Runner with Multiple Iterations for Stability\n\n# ========== DETERMINISTIC SETUP ==========\nimport os\nimport random\nimport numpy as np\n\n# Set environment variables BEFORE importing other libraries\nos.environ['PYTHONHASHSEED'] = '42'\nos.environ['OMP_NUM_THREADS'] = '1'\nos.environ['MKL_NUM_THREADS'] = '1'\nos.environ['OPENBLAS_NUM_THREADS'] = '1'\nos.environ['VECLIB_MAXIMUM_THREADS'] = '1'\nos.environ['NUMEXPR_NUM_THREADS'] = '1'\n\n# ========== CONFIGURATION - CHANGE THIS VALUE ==========\nEARLY_PERCENTAGE = 0.35  # Change this to 0.20, 0.25, 0.30, 0.35, 0.40, or 0.45\nN_ITERATIONS = 5  # Number of times to run the entire pipeline\n# ======================================================\n\n# Imports\nimport sys\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Feature Engineering\ndef feature_engineering(df):\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n    \n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    return df \n\n# Configuration\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    FEATURES = [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n        \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        \"buy_qty\", \"sell_qty\", \"volume\", \"X888\", \"X421\", \"X333\", \"X292\",\n    ]\n\n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    BASE_RANDOM_STATE = 42  # Base seed that will be modified for each iteration\n\ndef get_xgb_params(seed):\n    \"\"\"Get XGBoost parameters with specified seed\"\"\"\n    return {\n        'tree_method': 'hist', \n        'device': 'cpu',\n        'n_jobs': 1,\n        'colsample_bytree': 0.4111224922845363, \n        'colsample_bynode': 0.28869302181383194,\n        'gamma': 1.4665430311056709, \n        'learning_rate': 0.014053505540364681, \n        'max_depth': 7, \n        'max_leaves': 40, \n        'n_estimators': 500,\n        'reg_alpha': 27.791606770656145, \n        'reg_lambda': 84.90603428439086,\n        'subsample': 0.06567,\n        'verbosity': 0,\n        'random_state': seed,\n        'seed': seed,\n    }\n\n# Loading Data\ndef create_time_decay_weights(n: int, decay: float = 0.9, reverse: bool = False) -> np.ndarray:\n    \"\"\"Create time decay weights. If reverse=True, older data gets higher weight.\"\"\"\n    positions = np.arange(n)\n    if reverse:\n        normalized = 1.0 - (positions / (n - 1))\n    else:\n        normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()\n\ndef load_data():\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=Config.FEATURES + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=Config.FEATURES)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n\n    train_df = feature_engineering(train_df)\n    test_df = feature_engineering(test_df)\n    \n    print(f\"Loaded data - Train: {train_df.shape}, Test: {test_df.shape}, Submission: {submission_df.shape}\")\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\nConfig.FEATURES += [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\nConfig.FEATURES = list(set(Config.FEATURES))  # remove duplicates\n\n# Training and Evaluation\ndef get_model_slices(n_samples: int):\n    return [\n        {\"name\": \"full_data\", \"type\": \"full\", \"cutoff\": 0},\n        {\"name\": \"last_75pct\", \"type\": \"recent\", \"cutoff\": int(0.25 * n_samples)},\n        {\"name\": \"last_50pct\", \"type\": \"recent\", \"cutoff\": int(0.50 * n_samples)},\n        {\"name\": f\"first_{int(EARLY_PERCENTAGE*100)}pct\", \"type\": \"early\", \"cutoff\": int(EARLY_PERCENTAGE * n_samples)},\n    ]\n\ndef train_and_evaluate_single_iteration(train_df, test_df, iteration_seed):\n    \"\"\"Run a single iteration of training with a specific seed\"\"\"\n    print(f\"\\n{'='*60}\")\n    print(f\"Starting Iteration with seed {iteration_seed}\")\n    print(f\"{'='*60}\")\n    \n    # Set seeds for this iteration\n    np.random.seed(iteration_seed)\n    random.seed(iteration_seed)\n    \n    n_samples = len(train_df)\n    model_slices = get_model_slices(n_samples)\n\n    oof_preds = {\n        \"xgb\": {s[\"name\"]: np.zeros(n_samples) for s in model_slices}\n    }\n    test_preds = {\n        \"xgb\": {s[\"name\"]: np.zeros(len(test_df)) for s in model_slices}\n    }\n\n    full_weights = create_time_decay_weights(n_samples)\n    kf = KFold(n_splits=Config.N_FOLDS, shuffle=False)\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), start=1):\n        print(f\"\\n--- Fold {fold}/{Config.N_FOLDS} ---\")\n        X_valid = train_df.iloc[valid_idx][Config.FEATURES]\n        y_valid = train_df.iloc[valid_idx][Config.LABEL_COLUMN]\n\n        for s in model_slices:\n            cutoff = s[\"cutoff\"]\n            slice_name = s[\"name\"]\n            slice_type = s[\"type\"]\n            \n            if slice_type == \"full\":\n                subset = train_df.reset_index(drop=True)\n                rel_idx = train_idx\n                sw = full_weights[train_idx]\n                \n            elif slice_type == \"recent\":\n                subset = train_df.iloc[cutoff:].reset_index(drop=True)\n                rel_idx = train_idx[train_idx >= cutoff] - cutoff\n                if cutoff > 0:\n                    sw = create_time_decay_weights(len(subset))[rel_idx]\n                else:\n                    sw = full_weights[train_idx]\n                    \n            elif slice_type == \"early\":\n                subset = train_df.iloc[:cutoff].reset_index(drop=True)\n                rel_idx = train_idx[train_idx < cutoff]\n                if len(rel_idx) > 0:\n                    sw = create_time_decay_weights(len(subset))[rel_idx]\n                else:\n                    sw = np.array([])\n\n            if len(rel_idx) == 0:\n                print(f\"  Skipping slice: {slice_name} (no training data in fold)\")\n                continue\n\n            X_train = subset.iloc[rel_idx][Config.FEATURES]\n            y_train = subset.iloc[rel_idx][Config.LABEL_COLUMN]\n            \n            X_train_np = X_train.values\n            y_train_np = y_train.values\n            X_valid_np = X_valid.values\n            y_valid_np = y_valid.values\n            \n            print(f\"  Training slice: {slice_name}, samples: {len(X_train)}\")\n\n            # Create model with iteration-specific seed\n            model = XGBRegressor(**get_xgb_params(iteration_seed + fold))\n            model.fit(X_train_np, y_train_np, sample_weight=sw, \n                      eval_set=[(X_valid_np, y_valid_np)], verbose=False)\n            \n            # Handle predictions based on slice type\n            if slice_type == \"early\":\n                mask = valid_idx < cutoff\n                if mask.any():\n                    idxs = valid_idx[mask]\n                    oof_preds[\"xgb\"][slice_name][idxs] = model.predict(train_df.iloc[idxs][Config.FEATURES])\n                if (~mask).any():\n                    oof_preds[\"xgb\"][slice_name][valid_idx[~mask]] = oof_preds[\"xgb\"][\"full_data\"][valid_idx[~mask]]\n            else:\n                mask = valid_idx >= cutoff if slice_type == \"recent\" else np.ones(len(valid_idx), dtype=bool)\n                if mask.any():\n                    idxs = valid_idx[mask]\n                    oof_preds[\"xgb\"][slice_name][idxs] = model.predict(train_df.iloc[idxs][Config.FEATURES])\n                if slice_type == \"recent\" and cutoff > 0 and (~mask).any():\n                    oof_preds[\"xgb\"][slice_name][valid_idx[~mask]] = oof_preds[\"xgb\"][\"full_data\"][valid_idx[~mask]]\n\n            # Test predictions\n            test_preds[\"xgb\"][slice_name] += model.predict(test_df[Config.FEATURES])\n\n    # Normalize test predictions\n    for slice_name in test_preds[\"xgb\"]:\n        test_preds[\"xgb\"][slice_name] /= Config.N_FOLDS\n\n    return oof_preds, test_preds\n\ndef ensemble_single_iteration(train_df, oof_preds, test_preds):\n    \"\"\"Create ensemble for a single iteration\"\"\"\n    scores = {}\n    for s in oof_preds[\"xgb\"]:\n        score = pearsonr(train_df[Config.LABEL_COLUMN], oof_preds[\"xgb\"][s])[0]\n        scores[s] = score\n        print(f\"  xgb - {s}: {score:.4f}\")\n    \n    total_score = sum(scores.values())\n\n    # Simple average ensemble\n    oof_simple = np.mean(list(oof_preds[\"xgb\"].values()), axis=0)\n    test_simple = np.mean(list(test_preds[\"xgb\"].values()), axis=0)\n    score_simple = pearsonr(train_df[Config.LABEL_COLUMN], oof_simple)[0]\n\n    # Weighted ensemble based on OOF scores\n    oof_weighted = sum(scores[s] / total_score * oof_preds[\"xgb\"][s] for s in scores)\n    test_weighted = sum(scores[s] / total_score * test_preds[\"xgb\"][s] for s in scores)\n    score_weighted = pearsonr(train_df[Config.LABEL_COLUMN], oof_weighted)[0]\n\n    print(f\"\\nXGB Simple Ensemble Pearson:   {score_simple:.4f}\")\n    print(f\"XGB Weighted Ensemble Pearson: {score_weighted:.4f}\")\n\n    # Return the better performing ensemble\n    if score_weighted > score_simple:\n        return oof_weighted, test_weighted, score_weighted, \"weighted\"\n    else:\n        return oof_simple, test_simple, score_simple, \"simple\"\n\n# Main execution with multiple iterations\ndef train_with_multiple_iterations(train_df, test_df, submission_df):\n    \"\"\"Run multiple iterations and average the results\"\"\"\n    \n    all_oof_preds = []\n    all_test_preds = []\n    all_scores = []\n    all_types = []\n    \n    # Run multiple iterations with different seeds\n    for iter_num in range(N_ITERATIONS):\n        # Use different seed for each iteration\n        # You can also use the same seed if you prefer - XGBoost will still have some randomness\n        iteration_seed = Config.BASE_RANDOM_STATE + iter_num * 100\n        \n        print(f\"\\n\\n{'#'*80}\")\n        print(f\"# ITERATION {iter_num + 1} of {N_ITERATIONS}\")\n        print(f\"{'#'*80}\")\n        \n        # Run single iteration\n        oof_preds, test_preds = train_and_evaluate_single_iteration(train_df, test_df, iteration_seed)\n        \n        # Get ensemble for this iteration\n        oof_ensemble, test_ensemble, score, ensemble_type = ensemble_single_iteration(train_df, oof_preds, test_preds)\n        \n        all_oof_preds.append(oof_ensemble)\n        all_test_preds.append(test_ensemble)\n        all_scores.append(score)\n        all_types.append(ensemble_type)\n    \n    # Average all iterations\n    print(f\"\\n\\n{'='*80}\")\n    print(\"FINAL RESULTS - Averaging Across All Iterations\")\n    print(f\"{'='*80}\")\n    \n    print(f\"\\nIndividual iteration scores: {[f'{s:.4f}' for s in all_scores]}\")\n    print(f\"Ensemble types used: {all_types}\")\n    print(f\"Mean score: {np.mean(all_scores):.4f} ± {np.std(all_scores):.4f}\")\n    \n    # Calculate final predictions as average of all iterations\n    final_oof = np.mean(all_oof_preds, axis=0)\n    final_test = np.mean(all_test_preds, axis=0)\n    \n    # Calculate final score\n    final_score = pearsonr(train_df[Config.LABEL_COLUMN], final_oof)[0]\n    print(f\"\\nFINAL ensemble score (averaged predictions): {final_score:.4f}\")\n    \n    # Calculate variance of predictions to see stability\n    test_std = np.std(all_test_preds, axis=0)\n    print(f\"Average standard deviation of test predictions: {np.mean(test_std):.6f}\")\n    print(f\"Max standard deviation of test predictions: {np.max(test_std):.6f}\")\n    \n    # Save submission\n    filename = f\"submission_early_{int(EARLY_PERCENTAGE*100)}pct_{N_ITERATIONS}iter.csv\"\n    submission_df[\"prediction\"] = final_test\n    submission_df.to_csv(filename, index=False)\n    print(f\"\\nSaved: {filename}\")\n    \n    return final_score\n\n# Main\nif __name__ == \"__main__\":\n    print(f\"\\nRunning with EARLY_PERCENTAGE = {EARLY_PERCENTAGE} ({int(EARLY_PERCENTAGE*100)}%)\")\n    print(f\"Number of iterations: {N_ITERATIONS}\")\n    \n    train_df, test_df, submission_df = load_data()\n    final_score = train_with_multiple_iterations(train_df, test_df, submission_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}