{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-24T07:53:54.575764Z","iopub.execute_input":"2025-07-24T07:53:54.576033Z","iopub.status.idle":"2025-07-24T07:53:54.585024Z","shell.execute_reply.started":"2025-07-24T07:53:54.57601Z","shell.execute_reply":"2025-07-24T07:53:54.584406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom xgboost import XGBRegressor\nimport warnings\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T07:53:54.585692Z","iopub.execute_input":"2025-07-24T07:53:54.585874Z","iopub.status.idle":"2025-07-24T07:53:54.615975Z","shell.execute_reply.started":"2025-07-24T07:53:54.585859Z","shell.execute_reply":"2025-07-24T07:53:54.615053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SAMPLE_SUB_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    LABEL_COLUMN = \"label\"\n    RANDOM_STATE = 42\n    \n    # --- New Strategy: Time-Weighted Slices ---\n    # Each number represents the number of recent rows to train a model on\n    TIME_SLICES = [2_000_000, 1_000_000, 500_000]\n    \n    # Weights for the ensemble. Higher weight for models trained on more recent data.\n    ENSEMBLE_WEIGHTS = [0.2, 0.3, 0.5] # Must sum to 1.0 and match the order of TIME_SLICES\n    \n    # --- Using our Proven Feature Set ---\n    TOP_X_FEATURES = [\n        'X758', 'X752', 'X778', 'X425', 'X466', 'X780', 'X180', 'X611', 'X610', 'X769', \n        'X591', 'X757', 'X421', 'X751', 'X343', 'X445', 'X750', 'X344', 'X586', 'X607', \n        'X614', 'X501', 'X508', 'X779', 'X613', 'X772', 'X545', 'X499', 'X777', 'X682', \n        'X174', 'X97', 'X383', 'X136', 'X98', 'X427', 'X385', 'X179', 'X576', 'X341', \n        'X683', 'X768', 'X608', 'X96', 'X588', 'X582', 'X465', 'X444', 'X767'\n    ]\n    BASE_PUBLIC_FEATURES = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n    COLS_TO_LOAD = list(set(TOP_X_FEATURES + BASE_PUBLIC_FEATURES))\n    \n    # A single, powerful, well-regularized XGBoost configuration\n    XGB_PARAMS = {\n        'n_estimators': 1000, 'max_depth': 8, 'learning_rate': 0.01,\n        'subsample': 0.7, 'colsample_bytree': 0.7, 'reg_alpha': 0.1,\n        'reg_lambda': 0.1, 'min_child_weight': 20, 'gamma': 0.1,\n        'random_state': RANDOM_STATE, 'n_jobs': -1, \n        'verbosity': 0, 'device': 'gpu', 'tree_method': 'hist'\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T06:19:55.306936Z","iopub.execute_input":"2025-07-20T06:19:55.307154Z","iopub.status.idle":"2025-07-20T06:19:55.328154Z","shell.execute_reply.started":"2025-07-20T06:19:55.307128Z","shell.execute_reply":"2025-07-20T06:19:55.327668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    \"\"\"Creates a richer set of time-series and microstructure features.\"\"\"\n    df = df.copy()\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-8)\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['log_volume'] = np.log1p(df['volume'])\n    \n    windows = [50, 100, 200]\n    for window in windows:\n        df[f'ofi_mean_{window}'] = df['order_flow_imbalance'].rolling(window=window, min_periods=10).mean()\n        df[f'spread_std_{window}'] = df['bid_ask_spread'].rolling(window=window, min_periods=10).std()\n\n    df = df.replace([np.inf, -np.inf], np.nan)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T06:19:55.329357Z","iopub.execute_input":"2025-07-20T06:19:55.329581Z","iopub.status.idle":"2025-07-20T06:19:55.345759Z","shell.execute_reply.started":"2025-07-20T06:19:55.329557Z","shell.execute_reply":"2025-07-20T06:19:55.345245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    \n    # --- STAGE 1: Load Full Datasets ---\n    print(\"--- Stage 1: Loading Full Datasets ---\")\n    train_cols = Config.COLS_TO_LOAD + ['label']\n    test_cols = Config.COLS_TO_LOAD\n    \n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=train_cols)\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=test_cols)\n    gc.collect()\n\n    # --- STAGE 2: Time-Weighted Ensemble Training ---\n    print(\"\\n--- Stage 2: Training Time-Weighted Ensemble ---\")\n    all_test_predictions = []\n    \n    for i, slice_size in enumerate(Config.TIME_SLICES):\n        print(f\"  > Training model for recent {slice_size:,} rows...\")\n        \n        # Take the most recent slice of data\n        train_slice = train_df.tail(slice_size).copy()\n        \n        # Engineer features on the slice and the full test set\n        train_slice_fe = feature_engineering(train_slice)\n        test_fe = feature_engineering(test_df.copy()) # Use a copy to avoid re-engineering\n        \n        # Handle NaNs from rolling features by backfilling then filling with 0\n        train_slice_fe = train_slice_fe.fillna(method='bfill').fillna(0)\n        test_fe = test_fe.fillna(method='bfill').fillna(0)\n        \n        features = [col for col in test_fe.columns]\n        X_train_slice, y_train_slice = train_slice_fe[features], train_slice_fe[Config.LABEL_COLUMN]\n        X_test = test_fe[features]\n\n        # Train a single powerful model on this time slice\n        model = XGBRegressor(**Config.XGB_PARAMS)\n        model.fit(X_train_slice, y_train_slice)\n        \n        # Store predictions\n        all_test_predictions.append(model.predict(X_test))\n        \n        del train_slice, train_slice_fe, test_fe, X_train_slice, y_train_slice, X_test\n        gc.collect()\n\n    # --- STAGE 3: Generate Final Submission ---\n    print(\"\\n--- Stage 3: Applying Ensemble Weights and Saving Submission ---\")\n    \n    # Calculate the weighted average of predictions\n    final_prediction = np.zeros_like(all_test_predictions[0])\n    for weight, preds in zip(Config.ENSEMBLE_WEIGHTS, all_test_predictions):\n        final_prediction += weight * preds\n    \n    # Post-processing: Clip predictions\n    p01, p99 = np.percentile(train_df[Config.LABEL_COLUMN].dropna(), [1, 99])\n    final_prediction_clipped = np.clip(final_prediction, p01, p99)\n    \n    # Create submission file\n    submission_df = pd.read_csv(Config.SAMPLE_SUB_PATH)\n    if 'label' in submission_df.columns:\n        submission_df = submission_df.drop(columns=['label'])\n    submission_df[\"prediction\"] = final_prediction_clipped\n    submission_df.to_csv(\"submission.csv\", index=False)\n    \n    print(\"\\nSubmission file 'submission.csv' created successfully.\")\n    print(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T06:19:55.346339Z","iopub.execute_input":"2025-07-20T06:19:55.346557Z","iopub.status.idle":"2025-07-20T06:19:55.362506Z","shell.execute_reply.started":"2025-07-20T06:19:55.346535Z","shell.execute_reply":"2025-07-20T06:19:55.361883Z"}},"outputs":[],"execution_count":null}]}