{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Crypto Market Prediction ","metadata":{}},{"cell_type":"markdown","source":"**I'll implement a comprehensive solution incorporating crypto-specific features, advanced architectures, and robust validation strategies. Here's the complete implementation:**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom xgboost import XGBRegressor\nimport warnings\nimport gc\n\n# Suppress common warnings for a cleaner output\nwarnings.filterwarnings('ignore')\n\n# ==============================================================================\n# Configuration\n# ==============================================================================\nclass Config:\n    \"\"\"Holds all major configuration parameters for the pipeline.\"\"\"\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SAMPLE_SUB_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    LABEL_COLUMN = \"label\"\n    RANDOM_STATE = 42\n    \n    # Bootstrap settings for stability analysis\n    N_BOOTSTRAPS = 15         # Number of bootstrap models to train for a stable ensemble.\n    BOOTSTRAP_RATIO = 0.8     # Proportion of data to sample in each bootstrap.\n    \n    # A single, powerful, well-regularized XGBoost configuration\n    XGB_PARAMS = {\n        'n_estimators': 500, 'max_depth': 8, 'learning_rate': 0.02,\n        'subsample': 0.8, 'colsample_bytree': 0.8, 'reg_alpha': 0.5,\n        'reg_lambda': 0.5, 'min_child_weight': 10, 'gamma': 0.1,\n        'random_state': RANDOM_STATE, 'n_jobs': -1, \n        'verbosity': 0, 'device': 'gpu', 'tree_method': 'hist'\n    }\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T13:04:16.998627Z","iopub.execute_input":"2025-07-21T13:04:16.998846Z","iopub.status.idle":"2025-07-21T13:04:22.406011Z","shell.execute_reply.started":"2025-07-21T13:04:16.998828Z","shell.execute_reply":"2025-07-21T13:04:22.405459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# Feature Engineering\n# ==============================================================================\ndef feature_engineering(df):\n    \"\"\" Creates a rich set of features from the raw data. \"\"\"\n    df = df.copy()\n    \n    # 1. Basic Microstructure Ratios\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-8)\n    df['bid_ask_spread'] = (df['ask_qty'] - df['bid_qty'])\n    \n    # 2. Rolling Window Features for context\n    windows = [50, 200]\n    for window in windows:\n        df[f'ofi_mean_{window}'] = df['order_flow_imbalance'].rolling(window=window, min_periods=1).mean()\n        df[f'spread_std_{window}'] = df['bid_ask_spread'].rolling(window=window, min_periods=1).std()\n\n    # Handle any potential infinite values or NaNs created during engineering\n    df = df.replace([np.inf, -np.inf], np.nan).fillna(0)\n    return df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T13:07:56.197092Z","iopub.execute_input":"2025-07-21T13:07:56.197852Z","iopub.status.idle":"2025-07-21T13:07:56.203010Z","shell.execute_reply.started":"2025-07-21T13:07:56.197825Z","shell.execute_reply":"2025-07-21T13:07:56.202269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# Main Execution Pipeline\n# ==============================================================================\nif __name__ == \"__main__\":\n    \n    # --- STAGE 1: Load and Process Data Sequentially for Memory Efficiency ---\n    print(\"--- Stage 1: Loading and Processing Data ---\")\n    all_feature_cols = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"] + [f\"X{i}\" for i in range(1, 781)]\n    \n    # Process training data first\n    print(\"  Processing training data...\")\n    train = feature_engineering(pd.read_parquet(Config.TRAIN_PATH, columns=all_feature_cols + ['label']))\n    if '__index_level_0__' in train.columns:\n        train = train.rename(columns={'__index_level_0__': 'timestamp'}).sort_values('timestamp').reset_index(drop=True)\n    X, y = train.drop(columns=['label', '__index_level_0__', 'timestamp'], errors='ignore'), train[Config.LABEL_COLUMN]\n    \n    # Free up memory\n    del train\n    gc.collect()\n    \n    # Process test data second\n    print(\"  Processing test data...\")\n    test = feature_engineering(pd.read_parquet(Config.TEST_PATH, columns=all_feature_cols))\n    X_test = test[X.columns] # Ensure column order matches\n    \n    # Free up memory\n    del test\n    gc.collect()\n\n    # --- STAGE 2: Bootstrap Ensemble Training ---\n    print(f\"\\n--- Stage 2: Training Bootstrap Ensemble ({Config.N_BOOTSTRAPS} iterations) ---\")\n    \n    all_test_predictions = []\n    \n    for i in range(Config.N_BOOTSTRAPS):\n        print(f\"  > Running Bootstrap Iteration {i+1}/{Config.N_BOOTSTRAPS}...\")\n        \n        # Create a bootstrap sample using integer positions for .iloc\n        bootstrap_indices = np.random.choice(len(X), size=int(len(X) * Config.BOOTSTRAP_RATIO), replace=True)\n        X_boot, y_boot = X.iloc[bootstrap_indices], y.iloc[bootstrap_indices]\n        \n        # Train model on this bootstrap sample\n        params = Config.XGB_PARAMS.copy()\n        params['random_state'] = Config.RANDOM_STATE + i # Vary seed for each model\n        \n        model = XGBRegressor(**params)\n        model.fit(X_boot, y_boot)\n        \n        # Store predictions for the test set\n        all_test_predictions.append(model.predict(X_test))\n        \n        del model, X_boot, y_boot, bootstrap_indices\n        gc.collect()\n\n    # --- STAGE 3: Generate Final Submission ---\n    print(\"\\n--- Stage 3: Averaging Bootstrap Predictions and Saving Submission ---\")\n    \n    # The final prediction is the average of all predictions from all bootstrap runs\n    final_prediction = np.mean(all_test_predictions, axis=0)\n    \n    # Post-processing: Clip predictions to a reasonable range based on training labels\n    p01, p99 = np.percentile(y, [1, 99])\n    final_prediction_clipped = np.clip(final_prediction, p01, p99)\n    \n    # Create submission file\n    submission_df = pd.read_csv(Config.SAMPLE_SUB_PATH)\n    submission_df[\"prediction\"] = final_prediction_clipped\n    submission_df.to_csv(\"submission.csv\", index=False)\n    \n    print(\"\\nSubmission file 'submission.csv' created successfully.\")\n    print(\"This script is the final, recommended model for the competition.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T13:08:21.993463Z","iopub.execute_input":"2025-07-21T13:08:21.993781Z","iopub.status.idle":"2025-07-21T13:36:46.253585Z","shell.execute_reply.started":"2025-07-21T13:08:21.993759Z","shell.execute_reply":"2025-07-21T13:36:46.252867Z"}},"outputs":[],"execution_count":null}]}