{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\n# from xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\n\n\ndef min_max_rolling(df,col,rolling=30):\n    roll_min = df[col].rolling(rolling).min()\n    roll_max = df[col].rolling(rolling).max()\n    current = df[col]\n\n    return (current-roll_min)/(roll_max-roll_min)\n\ndef standardize_rolling(df,col,rolling=30):\n    roll_mean = df[col].rolling(rolling).mean()\n    roll_std = df[col].rolling(rolling).std()\n    current = df[col]\n\n    return (current-roll_mean)/(roll_std)\n\ndef feature_engineering(df):\n    # Stationary Features from Previous Version\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    df['normalized_net_flow'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']  # Temporary for calculations\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['kyle_lambda'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['realized_spread_proxy'] = 2 * np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = (df['buy_qty'] - df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['trade_informativeness'] = (df['buy_qty'] - df['sell_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = (df['buy_qty'] - df['sell_qty']) / (df['total_depth'] + 1e-10) * df['volume']\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (np.abs(df['bid_qty'] - df['ask_qty']) + 1e-10)\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume_depth_ratio'] * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['net_buying_ratio'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n\n    # Previous New Features\n    df['order_flow_momentum'] = df['order_flow_imbalance'].diff() / (df['volume'] + 1e-10)\n    df['volume_rolling_mean'] = df['volume'].rolling(window=5).mean()\n    df['relative_volume_change'] = (df['volume'] - df['volume_rolling_mean']) / (df['volume_rolling_mean'] + 1e-10)\n    df['spread_volatility'] = df['effective_spread_proxy'].rolling(window=5).std() / (df['effective_spread_proxy'].rolling(window=5).mean() + 1e-10)\n    df['depth_turnover'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['imbalance_persistence'] = df['order_flow_imbalance'].shift(1) / (df['order_flow_imbalance'] + 1e-10)\n    df['volume_diff'] = df['volume'].diff()\n    df['volume_acceleration'] = df['volume_diff'].diff() / (df['volume_rolling_mean'] + 1e-10)\n    df['normalized_price_impact'] = (df['buy_qty'] - df['sell_qty']) / (df['total_depth'] + df['volume'] + 1e-10)\n    df['liquidity_asymmetry'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n\n    # New Noise-Reducing Features\n    # 1. Smoothed Order Flow Imbalance (EMA)\n    df['smoothed_order_flow'] = df['order_flow_imbalance'].ewm(span=30, adjust=False).mean()\n    \n    # 2. Volatility-Adjusted Order Flow\n    df['order_flow_volatility'] = df['order_flow_imbalance'].rolling(window=30).std()\n    df['volatility_adjusted_flow'] = df['order_flow_imbalance'] / (df['order_flow_volatility'] + 1e-10)\n    \n    # 3. Rank-Based Volume (robust to outliers)\n    df['volume_rank'] = min_max_rolling(df,'volume',rolling=30)\n    \n    # 4. Lagged Bid-Ask Imbalance\n    df['lagged_bid_ask_imbalance'] = df['bid_ask_imbalance'].shift(1)\n    df['lagged_imbalance_change'] = (df['bid_ask_imbalance'] - df['lagged_bid_ask_imbalance']) / (df['total_depth'] + 1e-10)\n    \n    # 5. Volatility-Adjusted Spread\n    df['spread_volatility_rolling'] = df['effective_spread_proxy'].rolling(window=5).std()\n    df['volatility_adjusted_spread'] = df['effective_spread_proxy'] / (df['spread_volatility_rolling'] + 1e-10)\n    \n    # 6. Smoothed Depth Turnover (EMA)\n    df['smoothed_depth_turnover'] = df['depth_turnover'].ewm(span=30, adjust=False).mean()\n    \n    # 7. Relative Momentum of Net Buying Ratio\n    df['net_buying_momentum'] = df['net_buying_ratio'].diff() / (df['volume_rolling_mean'] + 1e-10)\n    \n    # 8. Robust Imbalance Ratio (using quantiles)\n    df['imbalance_quantile'] = df['order_flow_imbalance'].rolling(window=30).median()\n    df['robust_imbalance_ratio'] = df['order_flow_imbalance'] / (df['imbalance_quantile'] + 1e-10)\n\n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    # Drop temporary columns\n    df = df.drop(columns=['total_depth', 'volume_rolling_mean', 'volume_diff', \n                         'order_flow_volatility', 'spread_volatility_rolling', \n                         'lagged_bid_ask_imbalance','imbalance_quantile'], errors='ignore')\n    \n    return df\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold\nfrom catboost import CatBoostRegressor\nfrom scipy.stats import pearsonr\nimport pickle\nimport os\n\n# Configuration\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    FEATURES = ['X758', \"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\", \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X174\",\n                \"X178\", \"X168\", \"X612\", \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\" ]\n    SELECTED_FEATURES = [\n        'X758',    \n        # Base features (excluding non-stationary volume, bid_qty, ask_qty, buy_qty, sell_qty)\n        \"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\",\n        \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X174\", \"X178\", \"X168\", \"X612\", 'volume_rank',\n        # Engineered stationary features\n        \"buy_sell_ratio\", \"selling_pressure\", \"effective_spread_proxy\", \"bid_ask_imbalance\",\n        \"order_flow_imbalance\", \"liquidity_ratio\", \"normalized_net_flow\", \"buying_pressure\",\n        \"depth_imbalance\", \"relative_spread\", \"kyle_lambda\", \"flow_toxicity\",\n        \"aggressive_flow_ratio\", \"volume_depth_ratio\", \"activity_intensity\",\n        \"realized_spread_proxy\", \"price_impact_proxy\", \"quote_volatility_proxy\",\n        \"imbalance_volume_interaction\", \"trade_informativeness\", \"execution_shortfall_proxy\",\n        \"adverse_selection_proxy\", \"fill_probability\", \"execution_rate\", \"market_efficiency\",\n        \"imbalance_squared\", \"bid_ratio\", \"ask_ratio\", \"buy_ratio\", \"sell_ratio\",\n        \"liquidity_consumption\", \"market_stress\", \"depth_depletion\", \"net_buying_ratio\",\n        # Previous new features\n        \"order_flow_momentum\", \"relative_volume_change\", \"spread_volatility\",\n        \"depth_turnover\", \"imbalance_persistence\", \"volume_acceleration\",\n        \"normalized_price_impact\", \"liquidity_asymmetry\",\n        # New noise-reducing features\n        \"smoothed_order_flow\", \"volatility_adjusted_flow\", \n        \"lagged_imbalance_change\", \"volatility_adjusted_spread\", \"smoothed_depth_turnover\",\n        \"net_buying_momentum\", \"robust_imbalance_ratio\"\n    ]\n    \n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    RANDOM_STATE = 42\n    EMBARGO_SIZE = 100\n    CATBOOST_PARAMS = {\n\"learning_rate\": 0.03393830807147021,\n\"depth\": 4,\n\"l2_leaf_reg\": 7.158097238609412,\n\"iterations\": 936,\n\"subsample\": 0.5610191174223894,\n\"min_data_in_leaf\": 54,\n\"task_type\": \"GPU\",\n\"bootstrap_type\": \"Bernoulli\",\n\"random_seed\": 42,\n\"verbose\": 0\n}\n\ndef load_data():\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=Config.FEATURES + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=Config.FEATURES)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n\n    train_df = feature_engineering(train_df)\n    test_df = feature_engineering(test_df)\n    print(f\"Loaded data - Train: {train_df.shape}, Test: {test_df.shape}, Submission: {submission_df.shape}\")\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\ndef create_time_decay_weights(n: int, decay: float = 0.9) -> np.ndarray:\n    positions = np.arange(n)\n    normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()\n\ndef train_and_evaluate(train_df, test_df):\n    n_samples = len(train_df)\n    kf = KFold(n_splits=Config.N_FOLDS, shuffle=False)\n    oof_preds = np.zeros(n_samples)\n    test_preds = np.zeros(len(test_df))\n    trained_models = []\n\n    os.makedirs(\"models\", exist_ok=True)\n    full_weights = create_time_decay_weights(n_samples)\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), 1):\n        print(f\"Training Fold {fold}/{Config.N_FOLDS}\")\n\n        # Apply embargo\n        embargo_end = valid_idx[-1] + Config.EMBARGO_SIZE + 1 if valid_idx[-1] < n_samples - 1 else n_samples\n        embargo_mask = (train_idx < valid_idx[0]) | (train_idx >= embargo_end)\n        train_idx_embargoed = train_idx[embargo_mask]\n\n        # Prepare data\n        X_train = train_df.iloc[train_idx_embargoed][Config.SELECTED_FEATURES]\n        y_train = train_df.iloc[train_idx_embargoed][Config.LABEL_COLUMN]\n        X_valid = train_df.iloc[valid_idx][Config.SELECTED_FEATURES]\n        y_valid = train_df.iloc[valid_idx][Config.LABEL_COLUMN]\n        sample_weights = full_weights[train_idx_embargoed]\n\n        # Train model\n        model = CatBoostRegressor(**Config.CATBOOST_PARAMS)\n        model.fit(\n            X_train, y_train,\n            sample_weight=sample_weights,\n            eval_set=(X_valid, y_valid),\n            early_stopping_rounds=50\n        )\n\n        # Save model\n        model_path = f\"models/catboost_fold_{fold}.cbm\"\n        model.save_model(model_path)\n        trained_models.append(model_path)\n\n        # Out-of-fold predictions\n        oof_preds[valid_idx] = model.predict(X_valid)\n\n        # Test predictions\n        test_preds += model.predict(test_df[Config.SELECTED_FEATURES]) / Config.N_FOLDS\n\n        # Evaluate fold\n        fold_score = pearsonr(y_valid, oof_preds[valid_idx])[0]\n        print(f\"Fold {fold} Pearson Correlation: {fold_score:.4f}\")\n\n    oof_score = pearsonr(train_df[Config.LABEL_COLUMN], oof_preds)[0]\n    print(f\"\\nOverall OOF Pearson Correlation: {oof_score:.4f}\")\n\n    return oof_preds, test_preds, trained_models\n\ndef load_and_predict(test_df, model_paths):\n    test_preds = np.zeros(len(test_df))\n    for model_path in model_paths:\n        model = CatBoostRegressor()\n        model.load_model(model_path)\n        test_preds += model.predict(test_df[Config.SELECTED_FEATURES]) / len(model_paths)\n    return test_preds\n\ndef main():\n    # Load data (assuming feature_engineering is defined elsewhere)\n    train_df, test_df, submission_df = load_data()\n\n    print(f\"Train shape: {train_df.shape}, Test shape: {test_df.shape}\")\n\n    # Train and evaluate\n    oof_preds, test_preds, model_paths = train_and_evaluate(train_df, test_df)\n\n    # Generate submission\n    submission_df[\"prediction\"] = test_preds\n    submission_df.to_csv(\"submission.csv\", index=False)\n    print(\"Submission saved to submission.csv\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}