{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"a6a5a4fa-e795-43ec-949a-13408ee913e3","_cell_guid":"54444a80-4a5d-499d-8021-ad44b932453f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:50:49.242596Z","iopub.execute_input":"2025-07-24T20:50:49.242858Z","iopub.status.idle":"2025-07-24T20:50:49.507087Z","shell.execute_reply.started":"2025-07-24T20:50:49.242837Z","shell.execute_reply":"2025-07-24T20:50:49.506391Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom prophet import Prophet\nimport warnings\nwarnings.filterwarnings('ignore')\n\ndef simple_prophet_outlier_detection(df, feature_col, timestamp_col='__index_level_0__'):\n    \"\"\"\n    Simple example of using Prophet to detect outliers in a single feature.\n    \n    The idea: Prophet models the expected behavior of the feature over time.\n    Points that fall far outside Prophet's confidence interval are outliers.\n    \"\"\"\n    print(f\"Detecting outliers in {feature_col} using Prophet...\")\n    \n    # 1. Prepare data for Prophet\n    prophet_df = pd.DataFrame({\n        'ds': df[timestamp_col],\n        'y': df[feature_col]\n    })\n    \n    # Remove obvious bad values\n    prophet_df = prophet_df[np.isfinite(prophet_df['y'])]\n    \n    # Resample to hourly for efficiency (adjust based on your needs)\n    prophet_hourly = prophet_df.set_index('ds').resample('1H').mean().reset_index()\n    prophet_hourly = prophet_hourly.dropna()\n    \n    # 2. Fit Prophet model\n    model = Prophet(\n        changepoint_prior_scale=0.05,  # Low value = less sensitive to outliers\n        interval_width=0.95,  # 95% confidence interval\n        yearly_seasonality=False,\n        weekly_seasonality=True,\n        daily_seasonality=True\n    )\n    \n    model.fit(prophet_hourly)\n    \n    # 3. Generate predictions\n    forecast = model.predict(prophet_hourly)\n    \n    # 4. Identify outliers\n    # Method 1: Points outside confidence interval\n    outliers_ci = (\n        (prophet_hourly['y'] < forecast['yhat_lower']) | \n        (prophet_hourly['y'] > forecast['yhat_upper'])\n    )\n    \n    # Method 2: Points with large standardized residuals\n    residuals = prophet_hourly['y'] - forecast['yhat']\n    residual_std = residuals.std()\n    outliers_residual = np.abs(residuals) > 3 * residual_std\n    \n    # Combine both methods\n    outliers = outliers_ci & outliers_residual\n    \n    # 5. Visualize\n    fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 8))\n    \n    # Plot 1: Time series with outliers\n    ax1.plot(prophet_hourly['ds'], prophet_hourly['y'], 'b.', alpha=0.5, label='Actual')\n    ax1.plot(forecast['ds'], forecast['yhat'], 'g-', linewidth=2, label='Prophet Fit')\n    ax1.fill_between(forecast['ds'], forecast['yhat_lower'], forecast['yhat_upper'], \n                     alpha=0.2, color='green', label='95% CI')\n    \n    # Highlight outliers\n    outlier_points = prophet_hourly[outliers]\n    ax1.scatter(outlier_points['ds'], outlier_points['y'], \n               color='red', s=100, edgecolor='darkred', linewidth=2, \n               label=f'Outliers ({outliers.sum()})', zorder=10)\n    \n    ax1.set_title(f'Prophet Outlier Detection: {feature_col}')\n    ax1.set_xlabel('Date')\n    ax1.set_ylabel('Value')\n    ax1.legend()\n    ax1.grid(True, alpha=0.3)\n    \n    # Plot 2: Residual distribution\n    ax2.hist(residuals, bins=50, alpha=0.7, color='blue', edgecolor='black')\n    ax2.axvline(x=-3*residual_std, color='red', linestyle='--', label='±3σ threshold')\n    ax2.axvline(x=3*residual_std, color='red', linestyle='--')\n    ax2.set_title('Residual Distribution')\n    ax2.set_xlabel('Residual (Actual - Predicted)')\n    ax2.set_ylabel('Frequency')\n    ax2.legend()\n    ax2.grid(True, alpha=0.3)\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # 6. Return outlier information\n    outlier_info = {\n        'n_outliers': outliers.sum(),\n        'outlier_pct': outliers.sum() / len(prophet_hourly) * 100,\n        'outlier_timestamps': outlier_points['ds'].tolist(),\n        'outlier_values': outlier_points['y'].tolist(),\n        'residual_std': residual_std\n    }\n    \n    print(f\"Found {outlier_info['n_outliers']} outliers ({outlier_info['outlier_pct']:.2f}%)\")\n    \n    return outlier_info, model, forecast\n\n\n# Quick example for multiple features\ndef detect_outliers_multiple_features(data_path, features_to_check=None):\n    \"\"\"\n    Run outlier detection on multiple features.\n    \"\"\"\n    # Load data\n    if features_to_check is None:\n        # Default to some common features\n        features_to_check = ['volume', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty']\n    \n    df = pd.read_parquet(data_path, columns=['__index_level_0__'] + features_to_check)\n    \n    # Ensure timestamp\n    if '__index_level_0__' not in df.columns:\n        df['__index_level_0__'] = pd.date_range('2023-03-01', periods=len(df), freq='T')\n    \n    # Analyze each feature\n    outlier_summary = {}\n    \n    for feature in features_to_check:\n        if feature in df.columns:\n            print(f\"\\n{'='*60}\")\n            outlier_info, model, forecast = simple_prophet_outlier_detection(df, feature)\n            outlier_summary[feature] = outlier_info\n    \n    # Summary plot\n    fig, ax = plt.subplots(figsize=(10, 6))\n    \n    features = list(outlier_summary.keys())\n    outlier_pcts = [outlier_summary[f]['outlier_pct'] for f in features]\n    \n    bars = ax.bar(features, outlier_pcts, color='coral', edgecolor='darkred', linewidth=2)\n    \n    # Highlight high outlier features\n    for i, pct in enumerate(outlier_pcts):\n        if pct > 1.0:  # More than 1% outliers\n            bars[i].set_color('red')\n    \n    ax.set_title('Outlier Percentage by Feature', fontsize=14, fontweight='bold')\n    ax.set_ylabel('Outlier %')\n    ax.set_xlabel('Feature')\n    ax.grid(True, alpha=0.3, axis='y')\n    \n    # Add value labels\n    for bar, pct in zip(bars, outlier_pcts):\n        height = bar.get_height()\n        ax.text(bar.get_x() + bar.get_width()/2., height,\n                f'{pct:.2f}%', ha='center', va='bottom')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    return outlier_summary\n\n\n# Practical example: Using outlier detection for data cleaning\ndef clean_feature_outliers(df, feature_col, outlier_info, method='cap'):\n    \"\"\"\n    Clean outliers from a feature using different methods.\n    \"\"\"\n    # Get outlier timestamps\n    outlier_timestamps = outlier_info['outlier_timestamps']\n    \n    # Create copy\n    df_clean = df.copy()\n    \n    if method == 'remove':\n        # Remove rows with outliers\n        mask = ~df_clean['__index_level_0__'].isin(outlier_timestamps)\n        df_clean = df_clean[mask]\n        print(f\"Removed {len(outlier_timestamps)} rows with outliers\")\n        \n    elif method == 'cap':\n        # Cap outliers at 99th percentile\n        lower_cap = df_clean[feature_col].quantile(0.01)\n        upper_cap = df_clean[feature_col].quantile(0.99)\n        \n        df_clean[feature_col] = df_clean[feature_col].clip(lower=lower_cap, upper=upper_cap)\n        print(f\"Capped {feature_col} to range [{lower_cap:.2f}, {upper_cap:.2f}]\")\n        \n    elif method == 'interpolate':\n        # Replace outliers with interpolated values\n        outlier_mask = df_clean['__index_level_0__'].isin(outlier_timestamps)\n        df_clean.loc[outlier_mask, feature_col] = np.nan\n        df_clean[feature_col] = df_clean[feature_col].interpolate(method='linear')\n        print(f\"Interpolated {len(outlier_timestamps)} outlier values\")\n    \n    return df_clean\n\n\n# Define your feature list\n\nX_FEATURES = ['X363', 'X321', 'X405', 'X730', 'X523', 'X756', 'X589', 'X462', 'X779',\n                'X25', 'X532', 'X520', 'X329', 'X383', 'X751', 'X535', 'X639', 'X596', 'X761',\n            \"X752\",\n    \"X287\",\n    \"X298\",\n    \"X759\",\n    \"X302\",\n    \"X55\",\n    \"X56\",\n    \"X52\",\n    \"X303\",\n    \"X51\",\n    \"X598\", \"X385\", \"X603\", \"X674\",\n    \"X415\", \"X345\", \"X174\", \"X178\", \"X168\", \"X612\", \"bid_qty\",\n        \"ask_qty\", \"buy_qty\", \"sell_qty\" ]\n\n\n# Example usage\nif __name__ == \"__main__\":\n    # Example 1: Single feature analysis\n    data_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\n    df = pd.read_parquet(data_path, columns=['__index_level_0__', 'volume', 'label'] + X_FEATURES)\n    \n    if '__index_level_0__' not in df.columns:\n        df['__index_level_0__'] = pd.date_range('2023-03-01', periods=len(df), freq='T')\n    \n    # Detect outliers in volume\n    outlier_info, model, forecast = simple_prophet_outlier_detection(df, 'volume')\n    \n    # Example 2: Multiple features\n    outlier_summary = detect_outliers_multiple_features(\n        data_path,\n        X_FEATURES\n    )\n    \n    # Example 3: Clean the data\n    df_clean = df.copy()\n    for feat in X_FEATURES:\n        if feat in outlier_summary:\n            df_clean = clean_feature_outliers(df_clean, feat, outlier_summary[feat], method='cap')\n    print(\"\\nOutlier detection complete!\")","metadata":{"_uuid":"36a39d0c-8fd8-4d4d-a348-9d3d7af6eabf","_cell_guid":"5e30ab44-7bf3-4352-937e-9a7a6e3c5abc","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:50:49.508431Z","iopub.execute_input":"2025-07-24T20:50:49.508794Z","execution_failed":"2025-07-24T20:51:12.745Z"},"scrolled":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install koolbox scikit-learn==1.5.2","metadata":{"_uuid":"25deb7d0-a3db-4f48-8b4e-3f2441736b61","_cell_guid":"58cb6335-f5d0-4a5f-96eb-529492f23dde","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import Ridge\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr\nfrom xgboost import XGBRegressor\nfrom sklearn.base import clone\nfrom koolbox import Trainer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport optuna\nimport joblib\nimport gc\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"e634354c-1b08-42b7-b0db-5c6109c3b4c3","_cell_guid":"4ffee395-3269-401a-b54a-e33998ad976f","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    target = \"label\"\n    n_folds = 5\n    seed = 42\n\n    run_optuna = True\n    n_optuna_trials = 250","metadata":{"_uuid":"7d39e367-11e7-40e8-8f8a-4f9e9961f1c2","_cell_guid":"cbaab2e1-c04f-4525-b832-ad0feca43b16","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"_uuid":"a291af49-7b2a-4856-bd66-a21bc25142b3","_cell_guid":"82ef154b-870b-4bb7-9a32-fd703b1d20cb","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===== Feature Engineering =====\ndef feature_engineering(df):\n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # === NEW MICROSTRUCTURE FEATURES ===\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n        \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    # Handle infinities and NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    \n    # For each column, replace NaN with median for robustness\n    for col in df.columns:\n        if df[col].isna().any():\n            median_val = df[col].median()\n            df[col] = df[col].fillna(median_val if not pd.isna(median_val) else 0)\n    \n    return df","metadata":{"_uuid":"4f058407-ad9e-4d81-a4bd-c53b812bf790","_cell_guid":"5cd5858f-3e99-4a56-b899-1d493dd086b6","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = df_clean\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\n\nselected_columns = ['X363', 'X321', 'X405', 'X730', 'X523', 'X756', 'X589', 'X462', 'X779',\n                'X25', 'X532', 'X520', 'X329', 'X383', 'X751', 'X535', 'X639', 'X596', 'X761',\n            \"X752\",\n    \"X287\",\n    \"X298\",\n    \"X759\",\n    \"X302\",\n    \"X55\",\n    \"X56\",\n    \"X52\",\n    \"X303\",\n    \"X51\",\n    \"X598\", \"X385\", \"X603\", \"X674\",\n    \"X415\", \"X345\", \"X174\", \"X178\", \"X168\", \"X612\", \"bid_qty\",\n        \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\" ]\n\ntrain = train[selected_columns + [CFG.target]]\ntest = test[selected_columns]\n\n# Apply feature engineering\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\n\nto_remove = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n\ntrain = train.drop(columns=to_remove)\ntest = test.drop(columns=to_remove)\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nX = train.drop(CFG.target, axis=1)\ny = train[CFG.target]\n\nX_test = test","metadata":{"_uuid":"72647777-e55e-4ddf-bacb-b17a75a95e0e","_cell_guid":"911b2571-3975-4944-9534-cb5638776828","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"scrolled":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _pearsonr(y_true, y_pred):\n    return pearsonr(y_true, y_pred)[0]\n\nlgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.029312951475451557,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 126,\n    \"n_jobs\": -1,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.450669817684892,\n    \"verbose\": -1\n}\n\nlgbm_goss_params = {\n    \"boosting_type\": \"goss\",\n    \"colsample_bytree\": 0.34695458228489784,\n    \"learning_rate\": 0.031023014900595287,\n    \"min_child_samples\": 30,\n    \"min_child_weight\": 0.4727729225033618,\n    \"n_estimators\": 220,\n    \"n_jobs\": -1,\n    \"num_leaves\": 58,\n    \"random_state\": 42,\n    \"reg_alpha\": 38.665994901468224,\n    \"reg_lambda\": 92.76991677464294,\n    \"subsample\": 0.4810891284493255,\n    \"verbose\": -1\n}\n\nxgb_params = {\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}\n\nhistgb_params = {\n    \"max_iter\": 500,\n    \"learning_rate\": 0.05,\n    \"max_depth\": 10,\n    \"random_state\": 42\n}\n\ncatboost_params = {\n    \"iterations\": 1000,\n    \"learning_rate\": 0.02,\n    \"depth\": 10,\n    \"l2_leaf_reg\": 3,\n    \"bootstrap_type\": \"Bayesian\",\n    \"bagging_temperature\": 1.0,\n    \"loss_function\": \"RMSE\",\n    \"eval_metric\": \"RMSE\",\n    \"early_stopping_rounds\": 100,\n    \"random_seed\": 42,\n    \"task_type\": \"GPU\",\n    \"verbose\": 100\n}\n\n\nscores = {}\noof_preds = {}\ntest_preds = {}","metadata":{"_uuid":"89e50554-4db6-4075-b02f-8f1bc5b9cc5f","_cell_guid":"ddd3f503-dd2a-4361-b440-82a9fdc1ff3a","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_trainer = Trainer(\n    LGBMRegressor(**lgbm_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nlgbm_trainer.fit(X, y)\n\nscores[\"LightGBM (gbdt)\"] = lgbm_trainer.fold_scores\noof_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.oof_preds\ntest_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.predict(X_test)","metadata":{"_uuid":"02065873-28e7-49e5-beed-cc824ab4e891","_cell_guid":"6a9aabd6-db71-4e46-9eb2-76569efeaeb4","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.745Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_goss_trainer = Trainer(\n    LGBMRegressor(**lgbm_goss_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nlgbm_goss_trainer.fit(X, y)\n\nscores[\"LightGBM (goss)\"] = lgbm_goss_trainer.fold_scores\noof_preds[\"LightGBM (goss)\"] = lgbm_goss_trainer.oof_preds\ntest_preds[\"LightGBM (goss)\"] = lgbm_goss_trainer.predict(X_test)","metadata":{"_uuid":"18bc8a8d-efd2-4b39-89d1-6715cb0735ec","_cell_guid":"6db6948e-d208-48d3-bd91-a0f4ea9a8cc3","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_trainer = Trainer(\n    XGBRegressor(**xgb_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nxgb_trainer.fit(X, y)\n\nscores[\"XGBoost\"] = xgb_trainer.fold_scores\noof_preds[\"XGBoost\"] = xgb_trainer.oof_preds\ntest_preds[\"XGBoost\"] = xgb_trainer.predict(X_test)","metadata":{"_uuid":"8a9eba78-009b-49e1-9c29-40b42e253647","_cell_guid":"4c5c8a6b-d92e-4255-9ff4-b692d2a1e3b4","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"histgb_trainer = Trainer(\n    HistGradientBoostingRegressor(**histgb_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nhistgb_trainer.fit(X, y)\n\nscores[\"HistGB\"] = histgb_trainer.fold_scores\noof_preds[\"HistGB\"] = histgb_trainer.oof_preds\ntest_preds[\"HistGB\"] = histgb_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\ncatboost_trainer = Trainer(\n    CatBoostRegressor(**catboost_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\ncatboost_trainer.fit(X, y)\nscores[\"CatBoost\"] = catboost_trainer.fold_scores\noof_preds[\"CatBoost\"] = catboost_trainer.oof_preds\ntest_preds[\"CatBoost\"] = catboost_trainer.predict(X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_weights(weights, title):\n    sorted_indices = np.argsort(weights[0])[::-1]\n    sorted_coeffs = np.array(weights[0])[sorted_indices]\n    sorted_model_names = np.array(list(oof_preds.keys()))[sorted_indices]\n\n    plt.figure(figsize=(10, weights.shape[1] * 0.5))\n    ax = sns.barplot(x=sorted_coeffs, y=sorted_model_names, palette=\"RdYlGn_r\")\n\n    for i, (value, name) in enumerate(zip(sorted_coeffs, sorted_model_names)):\n        if value >= 0:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"left\", color=\"black\")\n        else:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"right\", color=\"black\")\n\n    xlim = ax.get_xlim()\n    ax.set_xlim(xlim[0] - 0.1 * abs(xlim[0]), xlim[1] + 0.1 * abs(xlim[1]))\n\n    plt.title(title)\n    plt.xlabel(\"\")\n    plt.ylabel(\"\")\n    plt.tight_layout()\n    plt.show()","metadata":{"_uuid":"264ae587-e8ef-4bb3-8663-0beb891304cb","_cell_guid":"448aef2c-45fc-4feb-b856-3118a38dc644","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = pd.DataFrame(oof_preds)\nX_test = pd.DataFrame(test_preds)","metadata":{"_uuid":"963a2d65-babf-46a6-9975-d5e4f0311243","_cell_guid":"8a378f7b-2651-4115-b9eb-c6581bc12fcf","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"joblib.dump(X, \"oof_preds.pkl\")\njoblib.dump(X_test, \"test_preds.pkl\")","metadata":{"_uuid":"b4dba65b-1cfa-48a5-ad0c-67564834c1fa","_cell_guid":"b33b9fb6-68bf-46fa-99e9-5f4fec8cb62f","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, RegressorMixin\nimport tensorflow as tf\nimport numpy as np\n\nclass AutoEncoderMLP(BaseEstimator, RegressorMixin):\n    def __init__(self, num_columns, hidden_units, dropout_rates, lr=1e-3):\n        self.num_columns = num_columns\n        self.hidden_units = hidden_units\n        self.dropout_rates = dropout_rates  # Should be [noise, dropout1, dropout2]\n        self.lr = lr\n        self.model = self._build_model()\n    \n    def _build_model(self):\n        inp = tf.keras.layers.Input(shape=(self.num_columns,))\n        x0 = tf.keras.layers.BatchNormalization()(inp)\n        \n        # Enhanced encoder with residual connections and attention\n        encoder = tf.keras.layers.GaussianNoise(self.dropout_rates[0])(x0)\n        \n        # First encoder layer\n        encoder1 = tf.keras.layers.Dense(self.hidden_units[0], activation='swish')(encoder)\n        encoder1 = tf.keras.layers.BatchNormalization()(encoder1)\n        encoder1 = tf.keras.layers.Dropout(self.dropout_rates[1])(encoder1)\n        \n        # Second encoder layer with residual connection\n        encoder2 = tf.keras.layers.Dense(self.hidden_units[0] // 2, activation='swish')(encoder1)\n        encoder2 = tf.keras.layers.BatchNormalization()(encoder2)\n        \n        # Bottleneck layer (compressed representation)\n        bottleneck = tf.keras.layers.Dense(self.hidden_units[0] // 4, activation='swish')(encoder2)\n        bottleneck = tf.keras.layers.BatchNormalization()(bottleneck)\n        \n        # Attention mechanism on bottleneck\n        attention_weights = tf.keras.layers.Dense(self.hidden_units[0] // 4, activation='sigmoid')(bottleneck)\n        attended_features = tf.keras.layers.Multiply()([bottleneck, attention_weights])\n        attended_features = tf.keras.layers.Dropout(self.dropout_rates[2])(attended_features)\n        \n        # Enhanced decoder path\n        decoder1 = tf.keras.layers.Dense(self.hidden_units[0] // 2, activation='swish')(attended_features)\n        decoder1 = tf.keras.layers.BatchNormalization()(decoder1)\n        \n        # Skip connection from encoder2 to decoder1\n        decoder1_skip = tf.keras.layers.Add()([decoder1, encoder2])\n        decoder1_skip = tf.keras.layers.Dropout(self.dropout_rates[1])(decoder1_skip)\n        \n        decoder2 = tf.keras.layers.Dense(self.hidden_units[0], activation='swish')(decoder1_skip)\n        decoder2 = tf.keras.layers.BatchNormalization()(decoder2)\n        \n        # Final decoder output\n        decoder_out = tf.keras.layers.Dense(self.num_columns, name='decoder')(decoder2)\n        \n        # Enhanced regression path with multiple branches\n        # Branch 1: Deep path\n        x_reg1 = tf.keras.layers.Dense(self.hidden_units[1], activation='swish')(attended_features)\n        x_reg1 = tf.keras.layers.BatchNormalization()(x_reg1)\n        x_reg1 = tf.keras.layers.Dropout(self.dropout_rates[2])(x_reg1)\n        \n        x_reg1 = tf.keras.layers.Dense(self.hidden_units[1] // 2, activation='swish')(x_reg1)\n        x_reg1 = tf.keras.layers.BatchNormalization()(x_reg1)\n        \n        # Branch 2: Wide path (direct from bottleneck)\n        x_reg2 = tf.keras.layers.Dense(self.hidden_units[1] // 2, activation='swish')(bottleneck)\n        x_reg2 = tf.keras.layers.BatchNormalization()(x_reg2)\n        \n        # Combine branches\n        x_reg_combined = tf.keras.layers.Concatenate()([x_reg1, x_reg2])\n        x_reg_combined = tf.keras.layers.Dropout(self.dropout_rates[2])(x_reg_combined)\n        \n        # Final regression layers\n        x_reg_final = tf.keras.layers.Dense(64, activation='swish')(x_reg_combined)\n        x_reg_final = tf.keras.layers.BatchNormalization()(x_reg_final)\n        x_reg_final = tf.keras.layers.Dropout(0.1)(x_reg_final)\n        \n        out_reg = tf.keras.layers.Dense(1, activation='linear', name='target')(x_reg_final)\n        \n        model = tf.keras.models.Model(inputs=inp, outputs=[decoder_out, out_reg])\n        \n        # Enhanced optimizer with scheduling\n        optimizer = tf.keras.optimizers.AdamW(\n            learning_rate=self.lr,\n            weight_decay=1e-4,  # L2 regularization\n            clipnorm=1.0  # Gradient clipping\n        )\n        \n        model.compile(\n            optimizer=optimizer,\n            loss={\n                \"decoder\": tf.keras.losses.Huber(delta=1.0),  # More robust to outliers\n                \"target\": tf.keras.losses.Huber(delta=1.0)\n            },\n            loss_weights={\"decoder\": 0.2, \"target\": 1.0}  # Reduced decoder weight\n        )\n        \n        return model\n\n    def fit(self, X, y):\n        self.model.fit(\n            X, {\"decoder\": X, \"target\": y},\n            epochs=50,\n            batch_size=8192,\n            validation_split=0.2,\n            callbacks=[\n                tf.keras.callbacks.EarlyStopping(patience=10, restore_best_weights=True),\n                tf.keras.callbacks.ReduceLROnPlateau(patience=5)\n            ],\n            verbose=0\n        )\n        return self\n\n    def predict(self, X):\n        _, y_pred = self.model.predict(X, verbose=0)\n        return y_pred.flatten()","metadata":{"_uuid":"b4a58234-22e8-436c-99ec-12a69f6bf6cb","_cell_guid":"bbeb9ea2-dd8e-4155-a64b-056cc0903706","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nae_model = AutoEncoderMLP(\n    num_columns=X.shape[1],\n    hidden_units=[128, 128],\n    dropout_rates=[0.06, 0.13, 0.23],\n    lr=1e-3\n)\n\n\nae_trainer = Trainer(\n    ae_model,\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=_pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nae_trainer.fit(X, y)\n\nscores[\"ae\"] = ae_trainer.fold_scores\noof_preds[\"ae\"] = ae_trainer.oof_preds\nae_test_preds= ae_trainer.predict(X_test)","metadata":{"_uuid":"c3c3be87-a55b-48bb-88fe-40c2216f6f25","_cell_guid":"01a301c5-00b3-44dc-8872-c1496f0f65aa","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(CFG.sample_sub_path)\nsub[\"prediction\"] = ae_test_preds\nsub.to_csv(f\"submission.csv\", index=False)\nsub.head()","metadata":{"_uuid":"36672853-daa3-485d-841c-ce1fdfffcf82","_cell_guid":"0ebf1950-3d7c-44f8-9b10-5210c7b1f9fa","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = pd.DataFrame(scores)\nmean_scores = scores.mean().sort_values(ascending=False)\norder = scores.mean().sort_values(ascending=False).index.tolist()\n\nmin_score = mean_scores.min()\nmax_score = mean_scores.max()\npadding = (max_score - min_score) * 0.5\nlower_limit = min_score - padding\nupper_limit = max_score + padding\n\nfig, axs = plt.subplots(1, 2, figsize=(15, scores.shape[1] * 0.5))\n\nboxplot = sns.boxplot(data=scores, order=order, ax=axs[0], orient=\"h\", color=\"grey\")\naxs[0].set_title(f\"Fold Score\")\naxs[0].set_xlabel(\"\")\naxs[0].set_ylabel(\"\")\n\nbarplot = sns.barplot(x=mean_scores.values, y=mean_scores.index, ax=axs[1], color=\"grey\")\naxs[1].set_title(f\"Average Score\")\naxs[1].set_xlabel(\"\")\naxs[1].set_xlim(left=lower_limit, right=upper_limit)\naxs[1].set_ylabel(\"\")\n\nfor i, (score, model) in enumerate(zip(mean_scores.values, mean_scores.index)):\n    color = \"cyan\" if \"ensemble\" in model.lower() else \"grey\"\n    barplot.patches[i].set_facecolor(color)\n    boxplot.patches[i].set_facecolor(color)\n    barplot.text(score, i, round(score, 6), va=\"center\")\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"66283366-d006-4fb9-99da-70273b192e8b","_cell_guid":"b29ef141-51f5-4a64-9ae1-18476977f6fb","trusted":true,"collapsed":false,"execution":{"execution_failed":"2025-07-24T20:51:12.746Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"_uuid":"ce3b4a3a-7863-4f67-9bf3-038f02129845","_cell_guid":"719a74c6-be2d-4836-9df7-720901cc65d9","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}