{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":96164,"databundleVersionId":12993472},{"sourceType":"datasetVersion","sourceId":12549077,"datasetId":1346,"databundleVersionId":13137365}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom datetime import datetime\nimport warnings\n\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n\nfrom scipy.stats import rankdata, pearsonr\n\nimport shap","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:36:51.407397Z","iopub.execute_input":"2025-07-23T21:36:51.407711Z","iopub.status.idle":"2025-07-23T21:37:03.744580Z","shell.execute_reply.started":"2025-07-23T21:36:51.407680Z","shell.execute_reply":"2025-07-23T21:37:03.743893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):\n    \"\"\"\n    Reduces the memory footprint of a DataFrame by downcasting numeric columns \n    to the most efficient data types that can safely hold the data.\n\n    This function iterates over each column in the DataFrame and attempts to:\n    - Downcast integer columns to the smallest possible int subtype (int8, int16, etc.).\n    - Downcast float columns to the smallest possible float subtype (float16, float32, etc.).\n    - Skip non-numeric columns.\n\n    It prints the memory usage before and after optimization, along with the percentage reduction.\n\n    Args:\n        dataframe (pd.DataFrame): The input DataFrame to optimize.\n        dataset (str): A label or name of the dataset for logging purposes.\n\n    Returns:\n        pd.DataFrame: The optimized DataFrame with reduced memory usage.\n    \"\"\"\n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.745326Z","iopub.execute_input":"2025-07-23T21:37:03.745719Z","iopub.status.idle":"2025-07-23T21:37:03.755719Z","shell.execute_reply.started":"2025-07-23T21:37:03.745696Z","shell.execute_reply":"2025-07-23T21:37:03.754821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_features(df):\n    df = df.copy()\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_liquidity'] + 1e-8)\n    df['normalized_spread'] = df['bid_ask_spread'] / (df['total_liquidity'] + 1e-8)\n    \n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['order_flow_imbalance'] = df['net_order_flow'] / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    df['volume_participation'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    df['aggressive_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-8)\n    \n    df['buy_pressure'] = df['buy_qty'] / (df['volume'] + 1e-8)\n    df['sell_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n    df['net_pressure'] = df['buy_pressure'] - df['sell_pressure']\n    df['pressure_ratio'] = df['buy_pressure'] / (df['sell_pressure'] + 1e-8)\n    \n    df['depth_ratio'] = df['total_liquidity'] / (df['volume'] + 1e-8)\n    df['bid_depth_ratio'] = df['bid_qty'] / (df['volume'] + 1e-8)\n    df['ask_depth_ratio'] = df['ask_qty'] / (df['volume'] + 1e-8)\n    df['depth_imbalance'] = (df['bid_depth_ratio'] - df['ask_depth_ratio']) / (df['depth_ratio'] + 1e-8)\n    \n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-8)\n    df['amihud_illiquidity'] = np.abs(df['net_pressure']) / (df['volume'] + 1e-8)\n    df['liquidity_consumption'] = df['volume'] / (df['total_liquidity'] + 1e-8)\n    \n    df['price_efficiency'] = 1 / (1 + df['amihud_illiquidity'])\n    df['execution_quality'] = df['volume'] / (df['bid_ask_spread'] + 1)\n    \n    df['pin_proxy'] = np.abs(df['order_flow_imbalance']) * df['amihud_illiquidity']\n    df['order_toxicity'] = np.abs(df['order_flow_imbalance']) * df['kyle_lambda']\n    \n    df['bid_momentum'] = df['bid_qty'] * df['buy_qty'] / (df['volume'] + 1e-8)\n    df['ask_momentum'] = df['ask_qty'] * df['sell_qty'] / (df['volume'] + 1e-8)\n    df['liquidity_adjusted_volume'] = df['volume'] / np.sqrt(df['total_liquidity'] + 1)\n    \n    df['log_volume'] = np.log1p(df['volume'])\n    df['log_liquidity'] = np.log1p(df['total_liquidity'])\n    df['log_spread'] = np.log1p(np.abs(df['bid_ask_spread']))\n\n    # Replace any NaNs or Infs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.757713Z","iopub.execute_input":"2025-07-23T21:37:03.757980Z","iopub.status.idle":"2025-07-23T21:37:03.783929Z","shell.execute_reply.started":"2025-07-23T21:37:03.757958Z","shell.execute_reply":"2025-07-23T21:37:03.783097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_statistical_features(df):\n    \"\"\"\n    Adds statistical aggregation features across all 'X'-prefixed columns:\n    - Mean\n    - Std Dev\n    - Range (max - min)\n    - Median\n    - 25th percentile\n    - 75th percentile\n    - Count of values above row mean\n    - Index of max and min column (numeric suffix)\n    - Cleans up NaNs and infs\n    \"\"\"\n    x_cols = [col for col in df.columns if col.startswith('X') and col[1:].isdigit()]\n    x_data = df[x_cols]\n\n    # Core stats\n    df['x_stat_mean'] = x_data.mean(axis=1)\n    df['x_stat_std'] = x_data.std(axis=1)\n    df['x_stat_range'] = x_data.max(axis=1) - x_data.min(axis=1)\n    df['x_stat_median'] = x_data.median(axis=1)\n    df['x_stat_p25'] = x_data.quantile(0.25, axis=1)\n    df['x_stat_p75'] = x_data.quantile(0.75, axis=1)\n\n    # Count of values above row mean\n    row_means = df['x_stat_mean'].values[:, None]\n    df['x_stat_above_mean_count'] = (x_data.values > row_means).sum(axis=1)\n\n    # Index (suffix) of max and min column\n    df['x_stat_idx_max'] = x_data.idxmax(axis=1).str.extract(r'(\\d+)', expand=False).astype(float)\n    df['x_stat_idx_min'] = x_data.idxmin(axis=1).str.extract(r'(\\d+)', expand=False).astype(float)\n\n    # Cleanup: handle infs and NaNs\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n    df.fillna(0, inplace=True)\n\n    return df\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.784745Z","iopub.execute_input":"2025-07-23T21:37:03.785099Z","iopub.status.idle":"2025-07-23T21:37:03.808029Z","shell.execute_reply.started":"2025-07-23T21:37:03.785067Z","shell.execute_reply":"2025-07-23T21:37:03.807045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_non_linear_x_market_interactions(df):\n    \"\"\"\n    Adds non-linear interaction features between 'X'-prefixed columns and selected market features.\n\n    - For each column starting with 'X' followed by digits (e.g., 'X1', 'X23'), creates new features by \n      multiplying the X-column with the logarithm of each available market feature.\n    - Market features considered: volume, buy_qty, sell_qty, x_stat_mean, x_stat_median.\n    - Only uses market features that are present in the DataFrame.\n    - Resulting feature name format: '{X_col}_log_x_{market_feature}'.\n    - Cleans the resulting DataFrame by replacing NaNs and infinities with zero.\n\n    Returns:\n        pd.DataFrame: The input DataFrame with new interaction features added.\n    \"\"\"\n    market_features = ['volume', 'buy_qty', 'sell_qty','x_stat_mean', 'x_stat_median']\n    x_cols = [col for col in df.columns if col.startswith('X') and col[1:].isdigit()]\n    \n    # Filter only available market features\n    market_features = [feat for feat in market_features if feat in df.columns]\n\n    for x_col in x_cols:\n        for m_feat in market_features:\n            new_col = f'{x_col}_log_x_{m_feat}'\n            df[new_col] = df[x_col] * np.log(df[m_feat])\n    \n    # Cleanup\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n    df.fillna(0, inplace=True)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.808919Z","iopub.execute_input":"2025-07-23T21:37:03.809195Z","iopub.status.idle":"2025-07-23T21:37:03.823410Z","shell.execute_reply.started":"2025-07-23T21:37:03.809171Z","shell.execute_reply":"2025-07-23T21:37:03.822523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def select_top_k_features_by_shap(X_train, y_train, feature_names, k=30):\n    \"\"\"Select top-k features using mean absolute SHAP value importance\"\"\"\n    print(\"Calculating SHAP-based feature importance...\")\n\n    # Convert categories to int\n    X_train = X_train.copy()\n    for col in X_train.select_dtypes(include='category').columns:\n        X_train[col] = X_train[col].astype(int)\n\n    # Train XGBoost model\n    params = {\n        'n_estimators': 100,\n        'max_depth': 6,\n        'learning_rate': 0.1,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'random_state': 42,\n        'n_jobs': -1,\n        'verbosity': 0\n    }\n\n    model = XGBRegressor(**params)\n    model.fit(X_train, y_train)\n\n    # Compute SHAP values\n    explainer = shap.Explainer(model, X_train)\n    shap_values = explainer(X_train)\n\n    # Calculate mean absolute SHAP value per feature\n    mean_abs_shap = np.abs(shap_values.values).mean(axis=0)\n    shap_importance_df = pd.DataFrame({\n        'feature': X_train.columns,\n        'importance': mean_abs_shap\n    }).sort_values('importance', ascending=False)\n\n    # Select top-k features\n    selected_features = shap_importance_df.head(k)['feature'].tolist()\n\n    # Always include critical features\n    critical_features = ['order_flow_imbalance', 'kyle_lambda', 'vpin', 'volume',\n                         'bid_ask_spread', 'liquidity_imbalance', 'buying_pressure']\n\n    for feat in critical_features:\n        if feat in feature_names and feat not in selected_features:\n            selected_features.append(feat)\n\n    print(f\"Selected top {k} SHAP features (plus critical ones if needed). Total: {len(selected_features)}\")\n\n    return selected_features, shap_importance_df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.824346Z","iopub.execute_input":"2025-07-23T21:37:03.824633Z","iopub.status.idle":"2025-07-23T21:37:03.844700Z","shell.execute_reply.started":"2025-07-23T21:37:03.824611Z","shell.execute_reply":"2025-07-23T21:37:03.843800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_time_decay_weights(n, decay=0.95):\n    \"\"\"\n    Generates a sequence of time-decay weights for a series of length `n`.\n\n    Weights are assigned such that more recent observations (towards the end of the series)\n    receive higher weight, following an exponential decay pattern. The weights are normalized\n    to sum to `n`, preserving the original scale.\n\n    Args:\n        n (int): Number of observations.\n        decay (float, optional): Decay rate between 0 and 1. Lower values decay faster.\n                                 Defaults to 0.95.\n\n    Returns:\n        np.ndarray: A 1D array of shape (n,) containing the normalized time-decay weights.\n    \"\"\"\n    pos = np.arange(n)\n    norm = pos / (n - 1)\n    w = decay ** (1.0 - norm)\n    return w * n / w.sum()\n\ndef adjust_weights_for_outliers(X, y, base_weights, outlier_fraction=0.001):\n    \"\"\"\n    Adjusts sample weights to downweight outliers based on model prediction residuals.\n\n    A quick Random Forest model is trained using the provided features and labels\n    to estimate residuals. Samples with the largest residuals (as defined by \n    `outlier_fraction`) are considered outliers and receive lower weights.\n\n    The adjustment scales down the base weights for outliers on a linear scale \n    between 0.2 and 0.8 of the original value, depending on residual severity.\n\n    Args:\n        X (pd.DataFrame or np.ndarray): Feature matrix.\n        y (pd.Series or np.ndarray): Target values.\n        base_weights (np.ndarray): Original sample weights.\n        outlier_fraction (float, optional): Fraction of samples to treat as outliers.\n                                            Defaults to 0.001.\n\n    Returns:\n        np.ndarray: Adjusted sample weights with reduced influence for outliers.\n    \"\"\"\n    # Train quick model to estimate residuals\n    model = RandomForestRegressor(n_estimators=50, max_depth=10, random_state=42, n_jobs=-1)\n    model.fit(X, y, sample_weight=base_weights)\n    preds = model.predict(X)\n    residuals = np.abs(y - preds)\n\n    # Top N residuals = outliers\n    n_outliers = max(1, int(outlier_fraction * len(residuals)))\n    threshold = np.partition(residuals, -n_outliers)[-n_outliers]\n    outlier_mask = residuals >= threshold\n\n    # Downweight outliers (linear scale: 0.2–0.8 of base weight)\n    adjusted_weights = base_weights.copy()\n    if outlier_mask.any():\n        res_out = residuals[outlier_mask]\n        res_norm = (res_out - res_out.min()) / (res_out.ptp() + 1e-8)\n        weight_factors = 0.8 - 0.6 * res_norm\n        adjusted_weights[outlier_mask] *= weight_factors\n\n    return adjusted_weights\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.845537Z","iopub.execute_input":"2025-07-23T21:37:03.845817Z","iopub.status.idle":"2025-07-23T21:37:03.865242Z","shell.execute_reply.started":"2025-07-23T21:37:03.845790Z","shell.execute_reply":"2025-07-23T21:37:03.864306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore')\n\ntrain = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ntest = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:03.867724Z","iopub.execute_input":"2025-07-23T21:37:03.868017Z","iopub.status.idle":"2025-07-23T21:37:47.903629Z","shell.execute_reply.started":"2025-07-23T21:37:03.867992Z","shell.execute_reply":"2025-07-23T21:37:47.902829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_features = list(dict.fromkeys([\n    \"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\",\n    \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X174\", \"X178\", \"X168\", \"X612\",\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\",\n    \"X758\", \"X296\", \"X611\", \"X780\", \"X451\", \"X25\", \"X591\", \"X727\", \"X427\", \"X288\",\n    \"X721\", \"X312\", \"X421\", \"X471\", \"X573\", \"X255\", \"X144\", \"X299\", \"X301\", \"X563\",\n    \"X737\", \"X702\", \"X507\", \"X306\", \"X501\", \"X586\", \"X43\", \"X517\", \"X248\", \"X137\",\n    \"X757\", \"X196\", \"X777\", \"X280\", \"X266\", \"X689\", \"X294\", \"X492\", \"X555\", \"X731\",\n    \"X262\", \"X576\", \"X13\", \"X518\", \"X502\", \"X558\", \"X6\", \"X602\", \"X695\", \"X703\",\n    \"X413\", \"X660\", \"X37\", \"X15\", \"X310\", \"X512\", \"X362\", \"X631\", \"X214\", \"X562\",\n    \"X488\", \"X510\", \"X256\", \"X35\", \"X128\", \"X86\", \"X170\", \"X30\", \"X265\", \"X323\",\n    \"X559\", \"X348\", \"X130\", \"X529\", \"X20\", \"X4\", \"X90\", \"X192\", \"X91\", \"X582\", \"X99\",\n    \"X24\", \"X317\", \"X707\", \"X653\", \"X519\", \"X557\", \"X371\", \"X84\", \"X83\", \"X360\",\n    \"X111\", \"X699\", \"X187\", \"X637\", \"X567\", \"X577\", \"X313\", \"X60\", \"X671\", \"X698\",\n    \"X701\", \"X725\", \"X292\", \"X638\", \"X741\", \"X379\", \"X700\", \"X614\", \"X676\", \"X516\",\n    \"X697\", \"X311\", \"X615\", \"X706\", \"X466\", \"X571\", \"X17\", \"X584\", \"X436\", \"X305\",\n    \"X34\", \"X282\", \"X681\", \"X7\", \"X208\", \"X41\", \"X536\", \"X548\", \"X776\", \"X87\", \"X40\",\n    \"X570\", \"X539\", \"X474\", \"X753\", \"X425\", \"X217\", \"X199\", \"X18\", \"X609\", \"X21\",\n    \"X277\", \"X279\", \"X326\", \"X540\", \"X688\", \"X553\", \"X452\", \"X738\", \"X183\",\n    \"label\"\n]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:47.904540Z","iopub.execute_input":"2025-07-23T21:37:47.904767Z","iopub.status.idle":"2025-07-23T21:37:47.912191Z","shell.execute_reply.started":"2025-07-23T21:37:47.904749Z","shell.execute_reply":"2025-07-23T21:37:47.911546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\ntrain = train[base_features]\ntest = test[base_features]\n\ntrain = train.dropna().reset_index(drop=True)\ntest = test.fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:47.913229Z","iopub.execute_input":"2025-07-23T21:37:47.913480Z","iopub.status.idle":"2025-07-23T21:37:59.920342Z","shell.execute_reply.started":"2025-07-23T21:37:47.913441Z","shell.execute_reply":"2025-07-23T21:37:59.919425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train = add_features(train)\nx_train = add_statistical_features(x_train)\nx_train = add_non_linear_x_market_interactions(x_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:37:59.921267Z","iopub.execute_input":"2025-07-23T21:37:59.921554Z","iopub.status.idle":"2025-07-23T21:39:14.809736Z","shell.execute_reply.started":"2025-07-23T21:37:59.921527Z","shell.execute_reply":"2025-07-23T21:39:14.808215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_test = add_features(test)\nx_test = add_statistical_features(x_test)\nx_test = add_non_linear_x_market_interactions(x_test)\nx_train = x_train[np.isfinite(x_train['label'])].reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:39:14.811738Z","iopub.execute_input":"2025-07-23T21:39:14.812080Z","iopub.status.idle":"2025-07-23T21:40:33.337745Z","shell.execute_reply.started":"2025-07-23T21:39:14.812051Z","shell.execute_reply":"2025-07-23T21:40:33.336830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure the index is a datetime type\nx_train.index = pd.to_datetime(x_train.index)\n\n# Sort the dataset by timestamp (index)\nx_train = x_train.sort_index()\n\n# Final training features and labels (no validation split)\nX_train = x_train.drop(columns=['label'])\ny_train = x_train['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:40:33.338812Z","iopub.execute_input":"2025-07-23T21:40:33.339127Z","iopub.status.idle":"2025-07-23T21:40:36.586844Z","shell.execute_reply.started":"2025-07-23T21:40:33.339101Z","shell.execute_reply":"2025-07-23T21:40:36.585933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_names = X_train.columns.tolist()\nselected_features, importance_df = select_top_k_features_by_shap(X_train, y_train, feature_names, k=80)\nX_train = X_train[selected_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T21:40:36.587810Z","iopub.execute_input":"2025-07-23T21:40:36.588125Z","iopub.status.idle":"2025-07-23T22:24:16.376209Z","shell.execute_reply.started":"2025-07-23T21:40:36.588099Z","shell.execute_reply":"2025-07-23T22:24:16.371976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"XGB_PARAMS = {\n    \"tree_method\": \"hist\",\n    \"device\": \"gpu\",\n    \"colsample_bylevel\": 0.4778,\n    \"colsample_bynode\": 0.3628,\n    \"colsample_bytree\": 0.7107,\n    \"gamma\": 1.7095,\n    \"learning_rate\": 0.02213,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"subsample\": 0.3,\n    \"reg_alpha\": 39.3524,\n    \"reg_lambda\": 75.4484,\n    \"verbosity\": 0,\n    \"random_state\": 42,\n    \"n_jobs\": -1\n}\n\nLEARNERS = [\n    {\"name\": \"xgb\", \"Estimator\": XGBRegressor, \"params\": XGB_PARAMS}\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T22:24:16.378935Z","iopub.execute_input":"2025-07-23T22:24:16.379990Z","iopub.status.idle":"2025-07-23T22:24:16.405085Z","shell.execute_reply.started":"2025-07-23T22:24:16.379855Z","shell.execute_reply":"2025-07-23T22:24:16.401446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure input is numpy and in correct time order\nX = X_train.values\ny = y_train.values\nn_samples = len(X)\n\n# Set up time-aware folds (no shuffling!)\nkf = KFold(n_splits=5, shuffle=False)\n\n# Initialize list for storing models\nmodels = []\n\n# Fold loop for ensembling (no need for OOF predictions)\nfor fold, (train_idx, _) in enumerate(kf.split(X)):\n    print(f\"\\nTraining fold {fold + 1}\")\n    \n    # Full training fold\n    X_tr, y_tr = X[train_idx], y[train_idx]\n    \n    # Time-decay + outlier-aware weights\n    time_decay = create_time_decay_weights(len(train_idx), decay=0.95)\n    weights = adjust_weights_for_outliers(X_tr, y_tr, time_decay)\n    \n    # Train model\n    model = XGBRegressor(**XGB_PARAMS)\n    model.fit(X_tr, y_tr, sample_weight=weights, verbose=False)\n    \n    # Save model\n    models.append(model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T22:24:16.408889Z","iopub.execute_input":"2025-07-23T22:24:16.411291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_names = [col for col in X_train.columns if col != 'label']\nx_test_aligned = x_test[feature_names]\nx_test_np = x_test_aligned.values if hasattr(x_test_aligned, \"values\") else x_test_aligned\ny_pred = np.mean([model.predict(x_test_np) for model in models], axis=0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\nsubmission[\"prediction\"] = y_pred\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"Submission file saved as 'submission.csv'\")\nsubmission.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}