{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-25T03:02:43.374092Z","iopub.execute_input":"2025-07-25T03:02:43.374341Z","iopub.status.idle":"2025-07-25T03:02:45.509558Z","shell.execute_reply.started":"2025-07-25T03:02:43.374321Z","shell.execute_reply":"2025-07-25T03:02:45.508734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom xgboost import XGBRegressor\nfrom sklearn.preprocessing import RobustScaler\nimport gc\n\nprint(\"Starting DRW Crypto Market Prediction - XGBoost Only Pipeline\")\nprint(\"=\" * 80)\n\n# Configuration\nTRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\nTEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\nSAMPLE_SUB_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n# Specified features\nSELECTED_FEATURES = [\n    'log_volume', 'bid_qty', 'X351', 'X103', 'X430', 'liquidity_imbalance', 'X621',\n    'X288', 'X749', 'buy_qty', 'X630', 'X434', 'X631', 'amihud_illiquidity', 'X660',\n    'X460', 'X459', 'X48', 'X494', 'X151', 'X257', 'X309', 'execution_quality', 'X456',\n    'X398', 'X356', 'X237', 'X483', 'X181', 'X611', 'X439', 'X472', 'X183', 'X51',\n    'ask_qty', 'X625', 'X232', 'X746', 'X145', 'X609', 'X138', 'X488', 'X109', 'X315',\n    'X236', 'sell_qty', 'X671', 'X478', 'X394', 'X466', 'X99', 'X282', 'X243',\n    'bid_ask_spread', 'X452', 'X620', 'X253', 'X666', 'X471', 'X431', 'X445', 'X388',\n    'X476', 'X18', 'X350', 'X54', 'X212', 'X400', 'X290', 'X22', 'X627', 'X233',\n    'X44', 'X357', 'X714', 'X696', 'X205', 'price_efficiency', 'X264', 'X213', 'X11',\n    'X473', 'X52', 'X219', 'kyle_lambda', 'X191', 'X190', 'X622', 'X490', 'X32',\n    'X368', 'X226', 'X700', 'X670', 'X141', 'X224', 'X235', 'X157', 'X487', 'X667',\n    'X57', 'volume', 'X198', 'X659', 'X286', 'X624', 'X238', 'X332', 'X46', 'X362',\n    'depth_ratio', 'X608', 'X56', 'X314', 'X10', 'X688', 'X753', 'X24', 'X259',\n    'X23', 'X747', 'X626', 'X97', 'X186', 'X770', 'X482', 'X693'\n]\n\ndef reduce_mem_usage(df, name=\"\"):\n    \"\"\"Optimize memory usage\"\"\"\n    print(f\"Optimizing memory for {name}...\")\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage: {start_mem:.2f} MB -> {end_mem:.2f} MB ({100*(start_mem-end_mem)/start_mem:.1f}% reduction)')\n    return df\n\ndef add_engineered_features(df):\n    \"\"\"Add the engineered features that are in the selected features list\"\"\"\n    eps = 1e-8\n    \n    # Only add features that are in our selected features list\n    if 'log_volume' in SELECTED_FEATURES and 'volume' in df.columns:\n        df['log_volume'] = np.log1p(df['volume'])\n    \n    if 'bid_ask_spread' in SELECTED_FEATURES and all(col in df.columns for col in ['ask_qty', 'bid_qty']):\n        df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    \n    if 'liquidity_imbalance' in SELECTED_FEATURES and all(col in df.columns for col in ['bid_qty', 'ask_qty']):\n        df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + eps)\n    \n    if 'kyle_lambda' in SELECTED_FEATURES and all(col in df.columns for col in ['buy_qty', 'sell_qty', 'volume']):\n        net_order_flow = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + eps)\n        sqrt_volume = np.sqrt(df['volume'])\n        df['kyle_lambda'] = net_order_flow / (sqrt_volume + eps)\n    \n    if 'amihud_illiquidity' in SELECTED_FEATURES and all(col in df.columns for col in ['buy_qty', 'sell_qty', 'volume']):\n        net_order_flow = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + eps)\n        log_volume = np.log1p(df['volume'])\n        df['amihud_illiquidity'] = np.abs(net_order_flow) / (log_volume + eps)\n    \n    if 'depth_ratio' in SELECTED_FEATURES and all(col in df.columns for col in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty']):\n        total_liquidity = df['bid_qty'] + df['ask_qty']\n        trade_volume = df['buy_qty'] + df['sell_qty']\n        df['depth_ratio'] = total_liquidity / (trade_volume + eps)\n    \n    # Add placeholders for features not in the data\n    if 'execution_quality' in SELECTED_FEATURES:\n        df['execution_quality'] = 0  # Placeholder\n    \n    if 'price_efficiency' in SELECTED_FEATURES:\n        df['price_efficiency'] = 0  # Placeholder\n    \n    # Handle infinities and NaNs\n    numeric_cols = df.select_dtypes(include=[np.number]).columns\n    for col in numeric_cols:\n        df[col] = df[col].replace([np.inf, -np.inf], [1e8, -1e8])\n        df[col] = df[col].fillna(0)\n        df[col] = np.clip(df[col], -1e8, 1e8)\n    \n    return df\n\ndef get_xgboost_models():\n    \"\"\"Get different XGBoost model configurations\"\"\"\n    models = {\n    \n        'xgb_standard_1': XGBRegressor(\n            n_estimators=1200,\n            learning_rate=0.01,\n            max_depth=8,\n            min_child_weight=3,\n            subsample=0.8,\n            colsample_bytree=0.8,\n            gamma=1,\n            reg_alpha=5,\n            reg_lambda=5,\n            random_state=42,\n            tree_method='hist',\n            verbosity=0\n        ),\n        'xgb_aggressive_1': XGBRegressor(\n            n_estimators=800,\n            learning_rate=0.02,\n            max_depth=12,\n            min_child_weight=1,\n            subsample=0.9,\n            colsample_bytree=0.9,\n            gamma=0.1,\n            reg_alpha=1,\n            reg_lambda=1,\n            random_state=44,\n            tree_method='hist',\n            verbosity=0\n        ),\n\n\n        'xgb_shallow': XGBRegressor(\n            n_estimators=2000,\n            learning_rate=0.005,\n            max_depth=4,\n            min_child_weight=10,\n            subsample=0.65,\n            colsample_bytree=0.65,\n            gamma=3,\n            reg_alpha=15,\n            reg_lambda=15,\n            random_state=47,\n            tree_method='hist',\n            verbosity=0\n        ),\n        'xgb_balanced': XGBRegressor(\n            n_estimators=1100,\n            learning_rate=0.012,\n            max_depth=9,\n            min_child_weight=4,\n            subsample=0.82,\n            colsample_bytree=0.82,\n            gamma=0.8,\n            reg_alpha=4,\n            reg_lambda=4,\n            random_state=48,\n            tree_method='hist',\n            verbosity=0\n        )\n    }\n    return models\n\ndef train_and_predict():\n    \"\"\"Main training and prediction pipeline\"\"\"\n    \n    # Load data\n    print(\"\\nLoading data...\")\n    train = pd.read_parquet(TRAIN_PATH)\n    test = pd.read_parquet(TEST_PATH)\n    submission = pd.read_csv(SAMPLE_SUB_PATH)\n    \n    print(f\"Train shape: {train.shape}\")\n    print(f\"Test shape: {test.shape}\")\n    \n    # Add engineered features\n    print(\"\\nAdding engineered features...\")\n    train = add_engineered_features(train)\n    test = add_engineered_features(test)\n    \n    # Memory optimization\n    train = reduce_mem_usage(train, \"train\")\n    test = reduce_mem_usage(test, \"test\")\n    \n    # Filter to only use features that exist in the data\n    available_features = [f for f in SELECTED_FEATURES if f in train.columns]\n    print(f\"\\nUsing {len(available_features)} features out of {len(SELECTED_FEATURES)} requested\")\n    \n    # Prepare data\n    X = train[available_features]\n    y = train['label']\n    X_test = test[available_features]\n    \n    # Scale features\n    print(\"\\nScaling features...\")\n    scaler = RobustScaler()\n    X_scaled = scaler.fit_transform(X)\n    X_test_scaled = scaler.transform(X_test)\n    \n    # Clip extreme values\n    X_scaled = np.clip(X_scaled, -10, 10)\n    X_test_scaled = np.clip(X_test_scaled, -10, 10)\n    \n    # Get models\n    models = get_xgboost_models()\n    test_predictions = []\n    \n    # Train each model on full training data\n    print(\"\\nTraining XGBoost models on full training data...\")\n    print(f\"Total models to train: {len(models)}\")\n    \n    for model_name, model in models.items():\n        print(f\"\\nTraining {model_name}...\")\n        print(f\"  Parameters: n_estimators={model.n_estimators}, \"\n              f\"learning_rate={model.learning_rate}, max_depth={model.max_depth}\")\n        \n        try:\n            # Train on full data\n            model.fit(X_scaled, y)\n            \n            # Predict on test data\n            test_pred = model.predict(X_test_scaled)\n            test_predictions.append(test_pred)\n            \n            print(f\"  {model_name} training completed successfully\")\n            \n        except Exception as e:\n            print(f\"  Error training {model_name}: {str(e)}\")\n            continue\n        \n        # Clean up memory\n        gc.collect()\n    \n    # Ensemble predictions\n    if len(test_predictions) > 0:\n        print(f\"\\nEnsembling {len(test_predictions)} XGBoost models...\")\n        \n        # Simple average ensemble\n        final_predictions = np.mean(test_predictions, axis=0)\n        \n        # Create submission\n        submission['prediction'] = final_predictions\n        submission.to_csv('submission.csv', index=False)\n        \n        print(\"\\nSubmission saved to submission.csv\")\n        print(\"\\nFirst 10 predictions:\")\n        print(submission.head(10))\n        \n        # Statistics\n        print(f\"\\nPrediction statistics:\")\n        print(f\"Mean: {final_predictions.mean():.4f}\")\n        print(f\"Std: {final_predictions.std():.4f}\")\n        print(f\"Min: {final_predictions.min():.4f}\")\n        print(f\"Max: {final_predictions.max():.4f}\")\n        print(f\"25th percentile: {np.percentile(final_predictions, 25):.4f}\")\n        print(f\"50th percentile (median): {np.percentile(final_predictions, 50):.4f}\")\n        print(f\"75th percentile: {np.percentile(final_predictions, 75):.4f}\")\n        \n    else:\n        print(\"\\nERROR: No models were successfully trained!\")\n\nif __name__ == \"__main__\":\n    train_and_predict()\n    print(\"\\nXGBoost-only pipeline completed!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-25T03:02:45.511199Z","iopub.execute_input":"2025-07-25T03:02:45.511620Z","iopub.status.idle":"2025-07-25T03:36:54.763958Z","shell.execute_reply.started":"2025-07-25T03:02:45.511599Z","shell.execute_reply":"2025-07-25T03:36:54.763179Z"}},"outputs":[],"execution_count":null}]}