{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### 🙏 Acknowledgements  \nCredit to all authors of previous notebooks and discussions — your work has been incredibly inspiring!\n\n---\n\n### 🔍 Key Observations\n\nHere are two things that stood out to me during this competition:\n\n1. **Data Length Matters**  \n   Models trained on the full dataset (or at least the most recent 90%) consistently outperform those trained on only the most recent 50%. It seems more historical context helps generalization.\n\n2. **Feature Interactions Can Be Powerful**  \n   Some features, when used jointly (not in isolation!), significantly boost performance. So exploring interactions is well worth the effort.\n\nI hope some of these findings can help you improve your score — good luck!\n\n---\n\n### ❓ Still a Work in Progress\n\nOne challenge I'm still facing:\n\n> I haven’t been able to design a solid cross-validation strategy that replicates the test set correlation reliably.  \n> If you’ve found a way to bridge this gap, I’d love to hear your thoughts!","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:19:17.250552Z","iopub.execute_input":"2025-06-29T16:19:17.250931Z","iopub.status.idle":"2025-06-29T16:19:18.656329Z","shell.execute_reply.started":"2025-06-29T16:19:17.250895Z","shell.execute_reply":"2025-06-29T16:19:18.655474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# Configuration\n# =========================\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    FEATURES = [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n        \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\", \"bid_qty\",\n        \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\", \"X888\", \"X421\", \"X333\",\"X817\", \n        \"X586\",  \"X292\"\n    ]\n\n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    RANDOM_STATE = 42\n\nXGB_PARAMS = {\n    \"tree_method\": \"hist\",\n    \"device\": \"gpu\",\n    \"colsample_bylevel\": 0.4778,\n    \"colsample_bynode\": 0.3628,\n    \"colsample_bytree\": 0.7107,\n    \"gamma\": 1.7095,\n    \"learning_rate\": 0.02213,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"subsample\": 0.06567,\n    \"reg_alpha\": 39.3524,\n    \"reg_lambda\": 75.4484,\n    \"verbosity\": 0,\n    \"random_state\": Config.RANDOM_STATE,\n    \"n_jobs\": -1\n}\n\nLEARNERS = [\n    {\"name\": \"xgb\", \"Estimator\": XGBRegressor, \"params\": XGB_PARAMS}\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:19:18.657864Z","iopub.execute_input":"2025-06-29T16:19:18.658391Z","iopub.status.idle":"2025-06-29T16:19:18.665631Z","shell.execute_reply.started":"2025-06-29T16:19:18.658359Z","shell.execute_reply":"2025-06-29T16:19:18.664712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_features(df):\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'])\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'])\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'])\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'])\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'])\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'])\n    \n\n    return df\n\ndef create_time_decay_weights(n: int, decay: float = 0.9) -> np.ndarray:\n    positions = np.arange(n)\n    normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()\n    \ndef load_data():\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=Config.FEATURES + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=Config.FEATURES)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n    print(f\"Loaded data - Train: {train_df.shape}, Test: {test_df.shape}, Submission: {submission_df.shape}\")\n\n    train_df = add_features(train_df)\n    test_df = add_features(test_df)\n\n    Config.FEATURES += [\"log_volume\", 'bid_ask_interaction', 'bid_buy_interaction', 'bid_sell_interaction', 'ask_buy_interaction',\n                        'ask_sell_interaction']\n\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\n\ndef get_model_slices(n_samples: int):\n    return [\n        {\"name\": \"full_data\", \"cutoff\": 0},\n        {\"name\": \"last_90pct\", \"cutoff\": int(0.10 * n_samples)},\n        {\"name\": \"last_85pct\", \"cutoff\": int(0.15 * n_samples)},\n        {\"name\": \"last_80pct\", \"cutoff\": int(0.20 * n_samples)},\n\n    ]\n\n\n# =========================\n# Training and Evaluation\n# =========================\ndef train_and_evaluate(train_df, test_df):\n    n_samples = len(train_df)\n    model_slices = get_model_slices(n_samples)\n\n    oof_preds = {\n        learner[\"name\"]: {s[\"name\"]: np.zeros(n_samples) for s in model_slices}\n        for learner in LEARNERS\n    }\n    test_preds = {\n        learner[\"name\"]: {s[\"name\"]: np.zeros(len(test_df)) for s in model_slices}\n        for learner in LEARNERS\n    }\n\n    full_weights = create_time_decay_weights(n_samples)\n    kf = KFold(n_splits=Config.N_FOLDS, shuffle=False)\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), start=1):\n        print(f\"\\n--- Fold {fold}/{Config.N_FOLDS} ---\")\n        X_valid = train_df.iloc[valid_idx][Config.FEATURES]\n        y_valid = train_df.iloc[valid_idx][Config.LABEL_COLUMN]\n\n        for s in model_slices:\n            cutoff = s[\"cutoff\"]\n            slice_name = s[\"name\"]\n            subset = train_df.iloc[cutoff:].reset_index(drop=True)\n            rel_idx = train_idx[train_idx >= cutoff] - cutoff\n\n            X_train = subset.iloc[rel_idx][Config.FEATURES]\n            y_train = subset.iloc[rel_idx][Config.LABEL_COLUMN]\n            sw = create_time_decay_weights(len(subset))[rel_idx] if cutoff > 0 else full_weights[train_idx]\n\n            print(f\"  Training slice: {slice_name}, samples: {len(X_train)}\")\n\n            for learner in LEARNERS:\n                model = learner[\"Estimator\"](**learner[\"params\"])\n                model.fit(X_train, y_train, sample_weight=sw, eval_set=[(X_valid, y_valid)], verbose=False)\n\n                mask = valid_idx >= cutoff\n                if mask.any():\n                    idxs = valid_idx[mask]\n                    oof_preds[learner[\"name\"]][slice_name][idxs] = model.predict(train_df.iloc[idxs][Config.FEATURES])\n                if cutoff > 0 and (~mask).any():\n                    oof_preds[learner[\"name\"]][slice_name][valid_idx[~mask]] = oof_preds[learner[\"name\"]][\"full_data\"][\n                        valid_idx[~mask]]\n\n\n                test_preds[learner[\"name\"]][slice_name] += model.predict(test_df[Config.FEATURES])\n\n    # Normalize test predictions\n    for learner_name in test_preds:\n        for slice_name in test_preds[learner_name]:\n            test_preds[learner_name][slice_name] /= (Config.N_FOLDS-1)\n\n    return oof_preds, test_preds, model_slices\n\n\n# =========================\n# Ensemble & Submission\n# =========================\ndef ensemble_and_submit(train_df, oof_preds, test_preds, submission_df):\n    learner_name = 'xgb'\n    weights = np.array([1,1,1,1])\n\n    oof_weighted = pd.DataFrame(oof_preds[learner_name]).values @ weights\n    test_weighted = pd.DataFrame(test_preds[learner_name]).values @ weights\n    score_weighted = pearsonr(train_df[Config.LABEL_COLUMN], oof_weighted)[0]\n    print(f\"{learner_name.upper()} Weighted Ensemble Pearson: {score_weighted:.4f}\")\n\n\n    submission_df[\"prediction\"] = test_weighted\n    submission_df.to_csv(\"submission.csv\", index=False)\n    print(\"Saved: submission.csv\")\n    print (submission_df.head(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:19:59.034653Z","iopub.execute_input":"2025-06-29T16:19:59.035467Z","iopub.status.idle":"2025-06-29T16:19:59.055920Z","shell.execute_reply.started":"2025-06-29T16:19:59.035430Z","shell.execute_reply":"2025-06-29T16:19:59.054893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    train_df, test_df, submission_df = load_data()\n    oof_preds, test_preds, model_slices = train_and_evaluate(train_df, test_df)\n    ensemble_and_submit(train_df, oof_preds, test_preds, submission_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:20:01.673844Z","iopub.execute_input":"2025-06-29T16:20:01.674911Z","iopub.status.idle":"2025-06-29T16:38:17.960047Z","shell.execute_reply.started":"2025-06-29T16:20:01.674827Z","shell.execute_reply":"2025-06-29T16:38:17.958945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}