{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":12561067,"sourceType":"datasetVersion","datasetId":7931713},{"sourceId":12562482,"sourceType":"datasetVersion","datasetId":7932816},{"sourceId":12565949,"sourceType":"datasetVersion","datasetId":7935320},{"sourceId":12566584,"sourceType":"datasetVersion","datasetId":7935780},{"sourceId":12570969,"sourceType":"datasetVersion","datasetId":7938670},{"sourceId":249869065,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport gc\nimport re\n\n\nfrom time import time\nfrom tqdm import tqdm\n\nimport optuna\n\nfrom scipy.stats import pearsonr, spearmanr\nfrom sklearn.decomposition import PCA\n\nfrom sklearn.model_selection import KFold, train_test_split\nfrom sklearn.metrics import make_scorer\n\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import SGDRegressor\n\nimport lightgbm as lgb\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import TensorDataset, DataLoader\n\nimport math\nfrom scipy.signal import savgol_filter\nfrom sklearn.mixture import GaussianMixture","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:14.804288Z","iopub.execute_input":"2025-07-24T19:09:14.804849Z","iopub.status.idle":"2025-07-24T19:09:19.786760Z","shell.execute_reply.started":"2025-07-24T19:09:14.804822Z","shell.execute_reply":"2025-07-24T19:09:19.785929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Competition Data Importation**\n## Loading","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ntest = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:19.787969Z","iopub.execute_input":"2025-07-24T19:09:19.788878Z","iopub.status.idle":"2025-07-24T19:09:36.031598Z","shell.execute_reply.started":"2025-07-24T19:09:19.788855Z","shell.execute_reply":"2025-07-24T19:09:36.030987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compressing","metadata":{}},{"cell_type":"code","source":"def compress_memory(df: pd.DataFrame) -> pd.DataFrame:\n    \n    mem_before = df.memory_usage(deep=True).sum() / 1024**2\n\n    for col in df.columns:\n        if col == 'label':\n            continue\n        col_type = df[col].dtype\n        if col_type == 'float64':\n            df[col] = df[col].astype('float32')\n        elif col_type == 'int64':\n            df[col] = df[col].astype('int32')\n\n    mem_after = df.memory_usage(deep=True).sum() / 1024**2\n\n    print(f\"Memory size before compression: {mem_before:.2f} MB\")\n    print(f\"Memory size after compression: {mem_after:.2f} MB\")\n    print(f\"Compression ratio: {mem_after / mem_before:.2%}\")\n\n    return df\n\ntrain = compress_memory(train)\ntest = compress_memory(test)\nprint(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:36.032330Z","iopub.execute_input":"2025-07-24T19:09:36.032604Z","iopub.status.idle":"2025-07-24T19:09:40.282558Z","shell.execute_reply.started":"2025-07-24T19:09:36.032579Z","shell.execute_reply":"2025-07-24T19:09:40.281738Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Alignment\nTo align trains by day, there are strictly 1440 lines per day (assuming one data per minute):\n* Generate a complete time index (from earliest to latest time, frequency 1 minute)\n* Reconstruct the train with this complete time index, and the missing time points automatically become NaN\n* Fill a column with fillna\n\n<span style=\"color:red\"> Do not do this step, after submitting the test, it will lead to poor results, presumably because the alignment filling method is not added.</span>","metadata":{}},{"cell_type":"code","source":"def daily_row_count_summary(df):\n    daily_counts = df.resample('D').size()\n    print(len(daily_counts))\n    count_of_days = daily_counts.value_counts().sort_index()\n    summary_df = pd.DataFrame([count_of_days.values], columns=count_of_days.index)\n    summary_df.index = ['Day Count']\n    print(summary_df)\n    # return summary_df\n\ndaily_row_count_summary(train)\n# print(len(test)) # 538150 -> 538150/1440 = 373.71527777777777\n\ndef align_train_fill_missing(train):\n    train = train.sort_index()\n\n    full_idx = pd.date_range(start=train.index.min(), end=train.index.max(), freq='T')\n\n    train_aligned = train.reindex(full_idx)\n\n    train_aligned = train_aligned.ffill().bfill()\n    \n    return train_aligned\n\n# print(\"Before alignment\", len(train))\n# train = align_train_fill_missing(train)\n# print(\"After alignment\", len(train)) # 527040 = 366*1440","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:40.284018Z","iopub.execute_input":"2025-07-24T19:09:40.284260Z","iopub.status.idle":"2025-07-24T19:09:41.272207Z","shell.execute_reply.started":"2025-07-24T19:09:40.284241Z","shell.execute_reply":"2025-07-24T19:09:41.271492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Time sequence sampling**\n\n* The sampling unit is \"one day\" and cannot be taken from the middle of the day (each block length is 1440 rows per day and aligned all day).\n* The total amount of sampling is collected according to \"days\". For example, continuous blocks are not required for 60-day data. As long as the sampling blocks are continuous within and between blocks, continuity is not required.\n* The sampling decays in time order, the closer to the tail (nearest) the greater the weight, and the sampling probability decreases.\n\n<span style=\"color:red\"> If it did not make sense to align the data, then it would not make sense here to keep the 1440 window to sample the data throughout the day. Therefore, the most recent data can be randomly selected for segmentation, and exponential decay can be tried later, but the window is not needed.</span>","metadata":{}},{"cell_type":"code","source":"def global_weighted_daily_sampling(train: pd.DataFrame,\n                                   day_rows: int = 1440,\n                                   total_days: int = 60,\n                                   decay_rate: float = 3):\n\n    n = len(train)\n    total_full_days = n // day_rows\n    if total_full_days < total_days:\n        raise ValueError(f\"The data is insufficient for sampling {total_days} day, only {total_full_days}day\")\n\n    day_start_indices = np.arange(total_full_days) * day_rows\n\n    distance_from_end = total_full_days - 1 - np.arange(total_full_days)\n\n    weights = np.exp(-decay_rate * distance_from_end)\n    weights /= weights.sum()\n\n    sampled_days = np.random.choice(total_full_days, size=total_days, replace=False, p=weights)\n\n    sampled_days_sorted = np.sort(sampled_days)\n\n    sampled_chunks = [train.iloc[start:start + day_rows].copy() for start in day_start_indices[sampled_days_sorted]]\n\n    sampled_df = pd.concat(sampled_chunks, axis=0).reset_index(drop=True)\n\n    return sampled_df\n\n# recently_train = global_weighted_daily_sampling(train, day_rows=1440, total_days=60, decay_rate=3)\n# print(recently_train.shape)  # (60 * 1440)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:41.272928Z","iopub.execute_input":"2025-07-24T19:09:41.273139Z","iopub.status.idle":"2025-07-24T19:09:41.279167Z","shell.execute_reply.started":"2025-07-24T19:09:41.273122Z","shell.execute_reply":"2025-07-24T19:09:41.278528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test Sorting","metadata":{}},{"cell_type":"code","source":"# Try to load precomputed timestamp reconstruction data\ntimestamp_recon_path = '/kaggle/input/the-order-of-the-test-rows-2/closest_rows.csv'\nuse_timestamp_reconstruction = os.path.exists(timestamp_recon_path)\n\nif use_timestamp_reconstruction:\n    print(\"Found timestamp reconstruction file, loading...\")\n    \n    # Load precomputed timestamp reconstruction data\n    t = pd.Series(pd.read_csv(timestamp_recon_path)['0'].to_numpy())\n    assert t.shape == (test.shape[0],)\n    print('Reconstructed timestamps share:', len(t[t >= 0]) / len(t))\n\n\n    # Process timestamp reconstruction\n    t -= 10080\n    t[t < 0] = 538149\n\n    t = t.sort_values()\n    t[t <= len(t)] = np.arange(t[t <= len(t)].shape[0])\n    t = t.sort_index()\n\n    t = pd.Series(np.arange(538150), index=t.to_numpy()).sort_index()\n\n\n    # Sort test dataset by reconstructed time order\n    test = test.iloc[t.to_numpy()]\n\nelse:\n    print(\"WARNING: Timestamp reconstruction file not found!\")\n    print(f\"Expected path: {timestamp_recon_path}\")\n    print(\"Proceeding without timestamp reconstruction...\")\n    print(\"This may significantly impact model performance since lagged features assume temporal order.\")\n    \n    t = pd.Series(np.arange(len(test)))\n\nprint(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:41.279966Z","iopub.execute_input":"2025-07-24T19:09:41.280314Z","iopub.status.idle":"2025-07-24T19:09:45.563897Z","shell.execute_reply.started":"2025-07-24T19:09:41.280296Z","shell.execute_reply":"2025-07-24T19:09:45.563159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_label = pd.DataFrame(index=test.index)\ntest_label[\"label\"] = np.nan\ntest_label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:45.564616Z","iopub.execute_input":"2025-07-24T19:09:45.564854Z","iopub.status.idle":"2025-07-24T19:09:45.576822Z","shell.execute_reply.started":"2025-07-24T19:09:45.564836Z","shell.execute_reply":"2025-07-24T19:09:45.575964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Non Anonymous Variable**\n\n<span style=\"color:red\"> Using only this processed Non Anonymous Variable, regardless of the model used, the score can be increased from **0.01 to 0.4+**. If \"Test Sorting\" is not used, CV can achieve the same improvement effect, but LB can only be increased to 0.02.</span>\n\n## Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    cols_to_normalize = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    scaler = StandardScaler()\n\n    df[cols_to_normalize] = scaler.fit_transform(df[cols_to_normalize])\n    \n    # === Basic interaction features ===\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    df['buy_sell_interaction'] = df['buy_qty'] * df['sell_qty']\n\n    # === Spread indicators ===\n    df['spread_indicator'] = (df['ask_qty'] - df['bid_qty']) / (df['ask_qty'] + df['bid_qty'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n\n    # === Volume-weighted features ===\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['volume_weighted_bid'] = df['bid_qty'] * df['volume']\n    df['volume_weighted_ask'] = df['ask_qty'] * df['volume']\n\n    # === Ratio features ===\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n\n    # === Buying and selling pressure indicators ===\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n\n    # === Liquidity measurements ===\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_liquidity'] + 1e-10)\n    df['depth_imbalance'] = df['liquidity_imbalance']  # reused name\n    df['total_depth'] = df['total_liquidity']  # reused name\n    df['log_depth'] = np.log1p(df['total_depth'])\n\n    # === Trade intensity and market activity ===\n    df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['avg_trade_size'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['volume_participation'] = (df['buy_qty'] + df['sell_qty']) / (df['total_liquidity'] + 1e-10)\n    df['market_activity'] = df['volume'] * df['total_liquidity']\n    df['activity_intensity'] = df['trade_intensity']  # reused name\n\n    # === Execution quality proxy indicators ===\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['realized_volatility_proxy'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['realized_spread_proxy'] = 2 * np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = (df['buy_qty'] - df['sell_qty']) / (df['total_depth'] + 1e-10)\n\n    # === Normalized volume features ===\n    df['normalized_buy_volume'] = df['buy_qty'] / (df['bid_qty'] + 1e-10)\n    df['normalized_sell_volume'] = df['sell_qty'] / (df['ask_qty'] + 1e-10)\n\n    # === Complex interaction features ===\n    df['liquidity_adjusted_imbalance'] = df['order_flow_imbalance'] * df['depth_ratio'] if 'depth_ratio' in df else df['order_flow_imbalance'] * (df['total_depth'] / (df['volume'] + 1e-10))\n    df['pressure_spread_interaction'] = df['buying_pressure'] * df['spread_indicator']\n    df['flow_depth_interaction'] = (df['buy_qty'] - df['sell_qty']) * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n\n    # === Information asymmetry and market efficiency indicators ===\n    df['trade_informativeness'] = (df['buy_qty'] - df['sell_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = ((df['buy_qty'] - df['sell_qty']) / (df['total_depth'] + 1e-10)) * df['volume']\n\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (np.abs(df['bid_qty'] - df['ask_qty']) + 1e-10)\n\n    # === Nonlinear transformations ===\n    df['log_volume'] = np.log1p(df['volume'])\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n\n    # === Relative ratio indicators ===\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n\n    # === Market pressure and stress indicators ===\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n\n    # === Directional indicators ===\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n\n    # --- Add advanced complex features ---\n    required_cols = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    if all(f in df.columns for f in required_cols):\n        bid = df['bid_qty'].values\n        ask = df['ask_qty'].values\n        buy = df['buy_qty'].values\n        sell = df['sell_qty'].values\n        vol = df['volume'].values\n\n        if 'kyle_lambda_complex' not in df.columns:\n            order_imbalance = (bid - ask) / (bid + ask + 1e-6)\n            flow_imbalance = (buy - sell) / (buy + sell + 1e-6)\n            kyle_lambda = flow_imbalance * np.sqrt(np.abs(order_imbalance)) / (np.log1p(vol) + 1e-6)\n            df['kyle_lambda_complex'] = kyle_lambda\n\n        if 'vol_adjusted_pressure' not in df.columns:\n            total_pressure = bid + ask\n            vol_adj_pressure = np.log1p(total_pressure) * np.exp(-vol / (vol.mean() + 1e-6))\n            df['vol_adjusted_pressure'] = vol_adj_pressure\n\n        if 'trade_intensity_asymmetry' not in df.columns:\n            buy_intensity = buy / (vol + 1e-6)\n            sell_intensity = sell / (vol + 1e-6)\n            intensity_asymmetry = np.sign(buy_intensity - sell_intensity) * np.log1p(np.abs(buy_intensity - sell_intensity))\n            df['trade_intensity_asymmetry'] = intensity_asymmetry\n\n        if 'bid_minus_ask' not in df.columns:\n            bid_ask_diff = bid - ask\n            df['bid_minus_ask'] = bid_ask_diff\n\n        if 'volume_gaussian_kernel' not in df.columns:\n            vol_kernel = np.exp(-((vol - vol.mean()) ** 2) / (2 * (vol.std() + 1e-6) ** 2))\n            df['volume_gaussian_kernel'] = vol_kernel\n\n    # === Replace infinite and missing values ===\n    df = df.replace([np.inf, -np.inf], np.nan).fillna(0)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:45.577659Z","iopub.execute_input":"2025-07-24T19:09:45.577944Z","iopub.status.idle":"2025-07-24T19:09:45.600192Z","shell.execute_reply.started":"2025-07-24T19:09:45.577921Z","shell.execute_reply":"2025-07-24T19:09:45.599504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = feature_engineering(train)\ntest = feature_engineering(test)\nprint(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:45.600958Z","iopub.execute_input":"2025-07-24T19:09:45.601195Z","iopub.status.idle":"2025-07-24T19:09:59.788494Z","shell.execute_reply.started":"2025-07-24T19:09:45.601168Z","shell.execute_reply":"2025-07-24T19:09:59.787739Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Outlier Processing\nIf the data distribution of anonymous variables and non-anonymous features is compared, it will be found that most of the anonymous variables are approximately normally distributed, while the non-anonymous feature distribution is often more extreme and unstable, and the correlation with label and feature importance score are also very insignificant. This step is to first make the distribution of non-anonymous features more gentle and filter out extreme data caused by calculation.","metadata":{}},{"cell_type":"code","source":"base_cols = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\nnon_anonymous_features = [\n    col for col in train.columns\n    if not col.startswith('X') and col != 'label' and col not in base_cols\n]\n\ncorrelations = train[non_anonymous_features].corrwith(train['label'])\ntop_5_features = correlations.abs().sort_values(ascending=False).head(5)\n\nprint(\"Top 5 features with the highest absolute correlation with label:\")\nfor feature, corr in top_5_features.items():\n    print(f\"{feature}: {corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:09:59.791096Z","iopub.execute_input":"2025-07-24T19:09:59.791598Z","iopub.status.idle":"2025-07-24T19:10:00.357987Z","shell.execute_reply.started":"2025-07-24T19:09:59.791579Z","shell.execute_reply":"2025-07-24T19:10:00.357163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def replace_extreme_outliers_with_quantiles(df, columns, top_n=30, lower_q=0.2, upper_q=0.8):\n    df_replaced = df.copy()\n    for col in columns:\n\n        lower_val = df[col].quantile(lower_q)\n        upper_val = df[col].quantile(upper_q)\n\n        min_indices = df[col].nsmallest(top_n).index\n        max_indices = df[col].nlargest(top_n).index\n\n        df_replaced.loc[min_indices, col] = lower_val\n\n        df_replaced.loc[max_indices, col] = upper_val\n\n    return df_replaced\n\ntrain_df = replace_extreme_outliers_with_quantiles(train, non_anonymous_features, top_n=20, lower_q=0.1, upper_q=0.9)\ntest_df = replace_extreme_outliers_with_quantiles(test, non_anonymous_features, top_n=20, lower_q=0.1, upper_q=0.9)\nprint(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:00.358898Z","iopub.execute_input":"2025-07-24T19:10:00.359274Z","iopub.status.idle":"2025-07-24T19:10:07.318813Z","shell.execute_reply.started":"2025-07-24T19:10:00.359247Z","shell.execute_reply":"2025-07-24T19:10:07.318002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_two_features_for_four_dfs(df1, df2, df3, df4):\n\n    dfs = [df1, df2, df3, df4]\n    titles = ['train_original', 'train_after', 'test_original', 'test_after']\n    features = ['normalized_buy_volume', 'bid_ask_ratio']\n    \n    for feature in features:\n        print(feature)\n        plt.figure(figsize=(16, 3))\n        for i, df in enumerate(dfs):\n            plt.subplot(1, 4, i + 1)\n            plt.plot(df[feature].values, color='steelblue', linewidth=0.5)\n            plt.title(titles[i], fontsize=10)\n            plt.xticks([])\n            plt.yticks([])\n        plt.tight_layout()\n        plt.show()\n\nplot_two_features_for_four_dfs(train, train_df, test, test_df)\n# del train, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:07.319539Z","iopub.execute_input":"2025-07-24T19:10:07.319811Z","iopub.status.idle":"2025-07-24T19:10:08.123730Z","shell.execute_reply.started":"2025-07-24T19:10:07.319781Z","shell.execute_reply":"2025-07-24T19:10:08.123035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlations = train_df[non_anonymous_features].corrwith(train_df['label'])\ntop_5_features = correlations.abs().sort_values(ascending=False).head(5)\n\nprint(\"Top 5 features with the highest absolute correlation with label:\")\nfor feature, corr in top_5_features.items():\n    print(f\"{feature}: {corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:08.124659Z","iopub.execute_input":"2025-07-24T19:10:08.124908Z","iopub.status.idle":"2025-07-24T19:10:08.716810Z","shell.execute_reply.started":"2025-07-24T19:10:08.124890Z","shell.execute_reply":"2025-07-24T19:10:08.715987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Denoising: Savitzky Golay\nThe purpose of this step is similar to that of the previous step, which is to make the data smoother and reduce the impact of extremes and noise. It is worth mentioning that the distribution of white noise is also normal, and \"savgol_filter\" can also filter out the trend in white noise when the window length is large enough, so it is a good tool. <span style=\"color:red\">It is worth noting that \"savgol_filter\" essentially uses future data, but this is allowed during the training phase. Moreover, we can see that the improvement of the effect is significantly improved in the correlation between the features and the label itself, which makes many features with low correlation and zero importance of the training features in the later stage meaningful. Therefore, this is very important in the application of time series prediction, although in this competition, we rely on the \"Test Sorting\" step, However, it is reasonable in the prediction of keeping time series in practice.</span>","metadata":{}},{"cell_type":"code","source":"def apply_savgol_filter(df, columns_to_invert):\n    \n    window_length = 721\n    polyorder = 3\n    for feature in columns_to_invert:\n        df[feature] = savgol_filter(df[feature], window_length=window_length, polyorder=polyorder)\n    window_length = 121\n    polyorder = 6\n    for feature in columns_to_invert:\n        df[feature] = savgol_filter(df[feature], window_length=window_length, polyorder=polyorder)\n    \n    return df\n\ntrain_df = apply_savgol_filter(train_df, non_anonymous_features)\ntest_df = apply_savgol_filter(test_df, non_anonymous_features)\n\ncorrelations = train_df[non_anonymous_features].corrwith(train_df['label'])\ntop_5_features = correlations.abs().sort_values(ascending=False).head(5)\n\nprint(train_df.shape)\nprint(test_df.shape)\n\nprint(\"Top 5 features with the highest absolute correlation with label:\")\nfor feature, corr in top_5_features.items():\n    print(f\"{feature}: {corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:08.717772Z","iopub.execute_input":"2025-07-24T19:10:08.718393Z","iopub.status.idle":"2025-07-24T19:10:54.830048Z","shell.execute_reply.started":"2025-07-24T19:10:08.718365Z","shell.execute_reply":"2025-07-24T19:10:54.829239Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Selection: Random Forest","metadata":{}},{"cell_type":"code","source":"def select_top_features_by_rf(df: pd.DataFrame, label_col: str = 'label', top_n: int = 10, n_estimators: int = 30) -> list:\n\n    X = df.drop(columns=[label_col])\n    y = df[label_col]\n\n    rf = RandomForestRegressor(n_estimators=n_estimators, random_state=42, n_jobs=-1)\n    rf.fit(X, y)\n\n    importances = rf.feature_importances_\n    feature_names = X.columns\n\n    imp_df = pd.DataFrame({\n        'feature': feature_names,\n        'importance': importances\n    })\n\n    imp_df = imp_df[imp_df['importance'] > 0]\n\n    imp_df = imp_df.sort_values(by='importance', ascending=False).head(top_n)\n\n    return imp_df['feature'].tolist()\n\n# recently_train = train_df[non_anonymous_features + [\"label\"]].tail(50000)\n# top_features = select_top_features_by_rf(recently_train, label_col='label', top_n=10, n_estimators=50)\n# print(\"Top features:\", top_features)\n# del recently_train\ntop_features = ['imbalance_volume_interaction', 'net_order_flow', 'bid_ask_interaction', 'normalized_sell_volume', 'buy_sell_ratio', 'sqrt_depth', 'signed_volume', 'imbalance_squared', 'bid_sell_interaction', 'relative_spread']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:54.830908Z","iopub.execute_input":"2025-07-24T19:10:54.831198Z","iopub.status.idle":"2025-07-24T19:10:54.836722Z","shell.execute_reply.started":"2025-07-24T19:10:54.831166Z","shell.execute_reply":"2025-07-24T19:10:54.835975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:54.837425Z","iopub.execute_input":"2025-07-24T19:10:54.837647Z","iopub.status.idle":"2025-07-24T19:10:55.013296Z","shell.execute_reply.started":"2025-07-24T19:10:54.837632Z","shell.execute_reply":"2025-07-24T19:10:55.012694Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Anonymous Variable**\n\n## Avoid Multicollinearity\n* Normalized MinMax(0-1)\n* Remove invalid variables: Delete variables with Unique less than 20\n* Filtered variables with low information content: variables with 20% of the variance removed\n* Remove collinear variables: those with corr greater than 0.8 delete the one with lower CORR (Spearman's correlation coefficient)","metadata":{}},{"cell_type":"code","source":"def process_anonymous_variables(\n    df,                         \n    label_col: str = \"label\",\n    anon_regex: str = r\"^X\\d+$\",\n    norm_eps: float = 1e-9,\n    uniq_thresh: int = 10,\n    collinear_thresh: float = 0.9,\n    var_quantile: float = 0.2,\n    verbose: bool = True,\n):\n    # --- 0. Ready Polars DataFrame ------------------------------------------\n    if isinstance(df, pl.DataFrame):\n        pl_df = df.clone()\n    else:\n        import pandas as pd\n        if not isinstance(df, pd.DataFrame):\n            raise TypeError(\"df Must be pandas.DataFrame or polars.DataFrame\")\n        pl_df = pl.from_pandas(df)\n\n    if label_col not in pl_df.columns:\n        raise ValueError(f\"column '{label_col}' doesn't in DataFrame\")\n\n    anon_cols = [c for c in pl_df.columns if re.match(anon_regex, c)]\n    if not anon_cols:\n        raise ValueError(\"No anonymous variables found\")\n    if verbose:\n        print(f\"The total number of original anonymous variables: {len(anon_cols)} col\")\n\n    # --- 1. Normalization ----------------------------------------------------------\n    pl_df = pl_df.with_columns([\n        ((pl.col(c) - pl.col(c).min()) /\n         (pl.col(c).max() - pl.col(c).min() + norm_eps)).alias(c)\n        for c in anon_cols\n    ])\n\n    # --- 2. Remove columns containing any infinity values ------------------\n    cols_to_keep = []\n    for c in anon_cols:\n        col_series = pl_df[c]\n        if col_series.is_infinite().any():\n            if verbose:\n                print(f\"Drop column with infinite values: {c}\")\n            continue\n        cols_to_keep.append(c)\n    anon_cols = cols_to_keep\n    if verbose:\n        print(f\"Step-1 After dropping infinite value columns, keep: {len(anon_cols)} col\")\n\n    # --- 3. Unique Filter -----------------------------------------------------\n    nuniques_raw = (\n        pl_df.select([pl.col(c).n_unique().alias(c) for c in anon_cols])\n        .to_dict(as_series=False)\n    )\n    nuniques = {k: (v[0] if isinstance(v, list) else v) for k, v in nuniques_raw.items()}\n    anon_cols = [c for c in anon_cols if nuniques[c] >= uniq_thresh]\n    if verbose:\n        print(f\"Step‑2 Drop unique < {uniq_thresh}, keep: {len(anon_cols)} col\")\n\n    # --- 4. Variance Filter --------------------------------------------------------\n    vars_raw = (\n        pl_df.select([pl.col(c).var().alias(c) for c in anon_cols])\n        .to_dict(as_series=False)\n    )\n    vars_dict = {k: (v[0] if isinstance(v, list) else v) for k, v in vars_raw.items()}\n    var_values = np.array([vars_dict[c] for c in anon_cols])\n    cutoff = np.quantile(var_values, var_quantile)\n    anon_cols = [c for c in anon_cols if vars_dict[c] > cutoff]\n    if verbose:\n        print(f\"Step‑3 Drop var lowest {var_quantile:.0%}, keep: {len(anon_cols)} col\")\n\n    # --- 5. NumPy --------------------------------------------------------\n    anon_mat = pl_df.select(anon_cols).to_numpy()\n    label_vals = pl_df.select(label_col).to_numpy().ravel()\n\n    # --- 6. Spearman Multicollinearity Filter -------------------------------------------\n    kept, dropped = [], set()\n    for i, col_i in tqdm(\n        enumerate(anon_cols),\n        total=len(anon_cols),\n        disable=not verbose,\n        desc=\"Step‑4 Spearman Multicollinearity Filter\",\n        dynamic_ncols=True,\n    ):\n        if col_i in dropped:\n            continue\n        kept.append(col_i)\n        xi = anon_mat[:, i]\n        for j in range(i + 1, len(anon_cols)):\n            col_j = anon_cols[j]\n            if col_j in dropped:\n                continue\n            rho, _ = spearmanr(xi, anon_mat[:, j], nan_policy=\"omit\")\n            if abs(rho) > collinear_thresh:\n                dropped.add(col_j)\n    anon_cols = kept\n    if verbose:\n        print(f\"Step‑4 Drop |ρ_s|>{collinear_thresh}, keep: {len(anon_cols)} col\")\n\n    # --- result -------------------------------------------------------\n    final_df = pl_df.to_pandas(use_pyarrow_extension_array=True)\n    kept_cols = anon_cols\n    other_cols = [c for c in final_df.columns if c not in kept_cols]\n    final_df = final_df[other_cols + kept_cols]\n\n    if verbose:\n        print(\"✅ The final retained anonymous variable columns:\", kept_cols)\n\n    return final_df\n\n# recently_train = train.tail(50000)\n# recently_train = process_anonymous_variables(recently_train, label_col=\"label\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:55.014246Z","iopub.execute_input":"2025-07-24T19:10:55.014461Z","iopub.status.idle":"2025-07-24T19:10:55.027786Z","shell.execute_reply.started":"2025-07-24T19:10:55.014444Z","shell.execute_reply":"2025-07-24T19:10:55.027138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Mutual Information Filtering\n* Variables with high degree of filtering overlap between X\n* Variables with high correlation between X and label are screened","metadata":{}},{"cell_type":"code","source":"def calc_mi_with_label(df, label_col='label', feature_prefix='X', eps=1e-9):\n    feature_cols = [col for col in df.columns if col.startswith(feature_prefix)]\n    \n    X = df[feature_cols].copy()\n    y = df[label_col]\n    \n    # Convert to float\n    try:\n        X_float = X.astype(float)\n    except Exception as e:\n        print(\"Failed to convert X to float:\", e)\n        raise\n    \n    # Replace positive and negative infinity with NaN\n    X_float.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n    # Fill NaNs forward then backward\n    X_float.fillna(method='ffill', inplace=True)\n    X_float.fillna(method='bfill', inplace=True)\n    \n    # Drop columns that still have NaNs\n    cols_with_nan = X_float.columns[X_float.isna().any()].tolist()\n    if cols_with_nan:\n        X_float.drop(columns=cols_with_nan, inplace=True)\n    else:\n        print(\"No NaNs remain after filling.\")\n\n    # Convert y to float and check NaNs\n    y_float = y.astype(float)\n    if y_float.isna().any():\n        raise ValueError(\"Input y contains NaN.\")\n\n    # Check for infinity values again\n    if np.isinf(X_float.to_numpy()).any():\n        raise ValueError(\"Input X contains infinite values.\")\n    if np.isinf(y_float.to_numpy()).any():\n        raise ValueError(\"Input y contains infinite values.\")\n\n    # Compute mutual information\n    mi_scores = mutual_info_regression(X_float.to_numpy(), y_float.to_numpy())\n    mi_series = pd.Series(mi_scores, index=X_float.columns).sort_values(ascending=False)\n\n    # Select columns with MI > 0.1\n    high_mi_cols = mi_series[mi_series > 0.1].index.tolist()\n    print(f\"Number of columns with mutual information > 0.1: {len(high_mi_cols)}\")\n    print(\"Columns with mutual information > 0.1:\", high_mi_cols)\n    \n    # Create final DataFrame with selected features + label\n    final_df = pd.concat([X_float[high_mi_cols], y_float], axis=1)\n    \n    return final_df, high_mi_cols\n\n\n# mi_recently_train, mi_list = calc_mi_with_label(recently_train, label_col='label')\n# mi_recently_train.head(3)\n# del recently_train, mi_recently_train\nmi_list = ['X613', 'X140', 'X182', 'X98', 'X345', 'X429', 'X387', 'X590', \n           'X584', 'X758', 'X769', 'X385', 'X610', 'X427', 'X428', 'X179', \n           'X780', 'X95', 'X137', 'X612', 'X178', 'X611', 'X344', 'X779', \n           'X386', 'X181', 'X608', 'X138', 'X587', 'X342', 'X134', 'X176', \n           'X92', 'X572', 'X578', 'X581', 'X609', 'X339', 'X381', 'X423', \n           'X384', 'X96', 'X180', 'X768', 'X301', 'X302', 'X300', 'X426', \n           'X422', 'X296', 'X379', 'X298', 'X303', 'X131', 'X94', 'X299', \n           'X380', 'X136', 'X219', 'X294', 'X767', 'X466', 'X343', 'X292', \n           'X295', 'X421', 'X569', 'X605', 'X416', 'X761', 'X297', 'X173', \n           'X293', 'X172', 'X606', 'X89', 'X336', 'X290', 'X175', 'X566', \n           'X575', 'X338', 'X88', 'X560', 'X90', 'X139', 'X132', 'X420', \n           'X378', 'X778', 'X291', 'X465', 'X288', 'X174', 'X86', 'X170', \n           'X128', 'X425', 'X628', 'X375', 'X417', 'X333', 'X289', 'X373', \n           'X383', 'X629', 'X130', 'X682', 'X445', 'X374', 'X683', 'X772', \n           'X585', 'X286', 'X218', 'X332', 'X762', 'X169', 'X563', 'X125', \n           'X586', 'X337', 'X607', 'X415', 'X557', 'X588', 'X655', 'X166', \n           'X627', 'X614', 'X82', 'X626', 'X592', 'X287', 'X766', 'X97', \n           'X372', 'X738', 'X126', 'X654', 'X341', 'X83', 'X133', 'X508', \n           'X410', 'X330', 'X583', 'X591', 'X589', 'X777', 'X739', 'X167', \n           'X377', 'X284', 'X419', 'X124', 'X678', 'X679', 'X414', 'X84', \n           'X444', 'X757', 'X127', 'X217', 'X464', 'X711', 'X335', 'X573', \n           'X163', 'X168', 'X582', 'X38', 'X501', 'X710', 'X580', 'X285', \n           'X684', 'X35', 'X91', 'X577', 'X574', 'X571', 'X331', 'X651', \n           'X282', 'X603', 'X656', 'X198', 'X602', 'X674', 'X283', 'X54', \n           'X650', 'X42', 'X368', 'X36', 'X56', 'X371', 'X624', 'X675', \n           'X41', 'X39', 'X554', 'X44', 'X37', 'X548', 'X40', 'X46', 'X750', \n           'X45', 'X765', 'X53', 'X121', 'X85', 'X625', 'X760', 'X740', \n           'X50', 'X247', 'X413', 'X773', 'X52', 'X47', 'X25', 'X48', 'X55', \n           'X226', 'X404', 'X49', 'X157', 'X367', 'X706', 'X240', 'X507', \n           'X576', 'X443', 'X452', 'X43', 'X707', 'X205', 'X272', 'X51', \n           'X33', 'X326', 'X570', 'X562', 'X451', 'X735', 'X369', 'X712', \n           'X411', 'X500', 'X561', 'X327', 'X450', 'X680', 'X329', 'X122', \n           'X164', 'X80', 'X734', 'X494', 'X559', 'X487', 'X646', 'X197', \n           'X764', 'X565', 'X473', 'X685', 'X160', 'X463']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:55.028701Z","iopub.execute_input":"2025-07-24T19:10:55.029014Z","iopub.status.idle":"2025-07-24T19:10:55.048903Z","shell.execute_reply.started":"2025-07-24T19:10:55.028991Z","shell.execute_reply":"2025-07-24T19:10:55.048304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = base_cols + top_features + mi_list + ['label']\ntest_cols = base_cols + top_features + mi_list\n\ntrain_cols = [col for col in train_cols if col in train_df.columns]\ntest_cols = [col for col in test_cols if col in test_df.columns]\n\ntrain_df = train_df[train_cols].copy()\ntest_df = test_df[test_cols].copy()\n\nprint(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:55.049604Z","iopub.execute_input":"2025-07-24T19:10:55.049851Z","iopub.status.idle":"2025-07-24T19:10:56.696052Z","shell.execute_reply.started":"2025-07-24T19:10:55.049824Z","shell.execute_reply":"2025-07-24T19:10:56.695323Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Dimension Reduction：PCA","metadata":{}},{"cell_type":"code","source":"def pca_on_anonymous_vars(train_data, test_data, feature_prefix='X', variance_threshold=0.95):\n    anon_cols = [col for col in train_data.columns if col.startswith(feature_prefix)]\n    non_anon_cols_train = [col for col in train_data.columns if not col.startswith(feature_prefix)]\n    non_anon_cols_test = [col for col in non_anon_cols_train if col in test_data.columns]\n    \n    X_train_anon = train_data[anon_cols].astype(float)\n    X_test_anon = test_data[anon_cols].astype(float)\n    \n    pca = PCA(n_components=variance_threshold)\n    X_train_pca = pca.fit_transform(X_train_anon)\n    \n    n_components = pca.n_components_\n    print(f\"Number of PCA components to retain {variance_threshold*100}% variance: {n_components}\")\n    \n    X_test_pca = pca.transform(X_test_anon)\n    \n    pca_cols = [f'PC{i+1}' for i in range(n_components)]\n    train_pca_data = pd.DataFrame(X_train_pca, columns=pca_cols, index=train_data.index)\n    test_pca_data = pd.DataFrame(X_test_pca, columns=pca_cols, index=test_data.index)\n    \n    train_final = pd.concat([train_data[non_anon_cols_train], train_pca_data], axis=1)\n    test_final = pd.concat([test_data[non_anon_cols_test], test_pca_data], axis=1)\n    \n    return n_components, train_final, test_final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:56.696932Z","iopub.execute_input":"2025-07-24T19:10:56.697188Z","iopub.status.idle":"2025-07-24T19:10:56.704579Z","shell.execute_reply.started":"2025-07-24T19:10:56.697167Z","shell.execute_reply":"2025-07-24T19:10:56.703705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_components, train_df, test_df = pca_on_anonymous_vars(train_df, test_df)\nprint(\"Components:\", n_components)\nprint(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:10:56.705514Z","iopub.execute_input":"2025-07-24T19:10:56.705754Z","iopub.status.idle":"2025-07-24T19:11:14.896840Z","shell.execute_reply.started":"2025-07-24T19:10:56.705737Z","shell.execute_reply":"2025-07-24T19:11:14.896153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trans_Anony = [col for col in train_df.columns if col.startswith(\"PC\")]\ncorrelations = train_df[trans_Anony].corrwith(train_df['label'])\ntop_5_features = correlations.abs().sort_values(ascending=False).head(5)\n\nprint(\"Top 5 features with the highest absolute correlation with label:\")\nfor feature, corr in top_5_features.items():\n    print(f\"{feature}: {corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:14.897534Z","iopub.execute_input":"2025-07-24T19:11:14.897804Z","iopub.status.idle":"2025-07-24T19:11:15.300382Z","shell.execute_reply.started":"2025-07-24T19:11:14.897775Z","shell.execute_reply":"2025-07-24T19:11:15.299654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# By lgbm importance:\npc_list = [\"PC2\", \"PC13\", \"PC5\", \"PC46\", \"PC27\", \"PC43\", \"PC22\", \"PC40\", \"PC48\", \"PC6\", \"PC28\", \"PC7\", \"PC20\", \"PC23\", \"PC33\", \"PC3\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:15.301061Z","iopub.execute_input":"2025-07-24T19:11:15.301268Z","iopub.status.idle":"2025-07-24T19:11:15.305261Z","shell.execute_reply.started":"2025-07-24T19:11:15.301251Z","shell.execute_reply":"2025-07-24T19:11:15.304566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **NN Encoder Features**\n\nTrain a fully connected neural network (MLP) on the original features, and extract compressed informative representations from the bottleneck layer—i.e., the layer just before the output layer, consisting of the top-level neurons. These bottleneck features are then used as new input features for subsequent models, such as LightGBM.","metadata":{}},{"cell_type":"code","source":"# print(train_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:15.305971Z","iopub.execute_input":"2025-07-24T19:11:15.306173Z","iopub.status.idle":"2025-07-24T19:11:15.317266Z","shell.execute_reply.started":"2025-07-24T19:11:15.306156Z","shell.execute_reply":"2025-07-24T19:11:15.316727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nn_input = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume',\n       'imbalance_volume_interaction', 'net_order_flow', 'bid_ask_interaction',\n       'normalized_sell_volume', 'buy_sell_ratio', 'sqrt_depth',\n       'signed_volume', 'imbalance_squared', 'bid_sell_interaction',\n       'relative_spread', 'PC1', 'PC2', 'PC3', 'PC4', 'PC5', 'PC6',\n       'PC7', 'PC8', 'PC9', 'PC10', 'PC11', 'PC12', 'PC13', 'PC14', 'PC15',\n       'PC16', 'PC17', 'PC18', 'PC19', 'PC20', 'PC21', 'PC22', 'PC23', 'PC24',\n       'PC25', 'PC26', 'PC27', 'PC28', 'PC29', 'PC30', 'PC31', 'PC32', 'PC33',\n       'PC34', 'PC35', 'PC36', 'PC37', 'PC38', 'PC39', 'PC40', 'PC41', 'PC42',\n       'PC43', 'PC44', 'PC45', 'PC46', 'PC47', 'PC48', 'PC49', 'PC50']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:15.317979Z","iopub.execute_input":"2025-07-24T19:11:15.318173Z","iopub.status.idle":"2025-07-24T19:11:15.328881Z","shell.execute_reply.started":"2025-07-24T19:11:15.318158Z","shell.execute_reply":"2025-07-24T19:11:15.328304Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Activation Function: Relu + Swish","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\nX_train_full = train_df[nn_input].tail(420000).values.astype(np.float32)\ny_train_full = train_df[\"label\"].tail(420000).values.astype(np.float32).reshape(-1, 1)\nX_test = test_df[nn_input].values.astype(np.float32)\n\nsplit_idx = int(len(X_train_full) * 0.8)\nX_train = X_train_full[:split_idx]\ny_train = y_train_full[:split_idx]\nX_val = X_train_full[split_idx:]\ny_val = y_train_full[split_idx:]\n\n# DataLoader\ntrain_dataset = TensorDataset(torch.from_numpy(X_train), torch.from_numpy(y_train))\nval_dataset = TensorDataset(torch.from_numpy(X_val), torch.from_numpy(y_val))\n\ntrain_loader = DataLoader(train_dataset, batch_size=128, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=128, shuffle=False)\n\n# Pearson Loss\ndef pearson_loss(pred, target):\n    pred_mean = torch.mean(pred)\n    target_mean = torch.mean(target)\n    pred_cent = pred - pred_mean\n    target_cent = target - target_mean\n    cov = torch.mean(pred_cent * target_cent)\n    pred_var = torch.mean(pred_cent ** 2)\n    target_var = torch.mean(target_cent ** 2)\n    corr = cov / (torch.sqrt(pred_var) * torch.sqrt(target_var) + 1e-8)\n    return 1 - corr\n\n# MLP Model\nclass FeedForwardBottleneck(nn.Module):\n    def __init__(self, input_dim=X_train.shape[1], dropout=0.3):\n        super().__init__()\n        self.layer1 = nn.Linear(input_dim, 256)\n        self.layer2 = nn.Linear(256, 128)\n        self.layer3 = nn.Linear(128, 64)\n        self.layer4 = nn.Linear(64, 32)\n        self.bottleneck = nn.Linear(32, 16)\n        self.regressor = nn.Linear(16, 1)\n        self.dropout = nn.Dropout(dropout)\n\n    def forward(self, x):\n        x = F.relu(self.layer1(x))\n        x = self.dropout(x)\n        x = F.silu(self.layer2(x))\n        x = self.dropout(x)\n        x = F.silu(self.layer3(x))\n        x = self.dropout(x)\n        x = F.silu(self.layer4(x))\n        x = self.dropout(x)\n        z = F.silu(self.bottleneck(x))\n        out = self.regressor(z)\n        return out, z\n\n# EarlyStopping\nclass EarlyStopping:\n    def __init__(self, patience=5, min_delta=1e-4):\n        self.patience = patience\n        self.min_delta = min_delta\n        self.counter = 0\n        self.best_loss = None\n        self.early_stop = False\n\n    def __call__(self, val_loss):\n        if self.best_loss is None or val_loss < self.best_loss - self.min_delta:\n            self.best_loss = val_loss\n            self.counter = 0\n        else:\n            self.counter += 1\n            print(f\"⚠️ EarlyStopping counter: {self.counter}/{self.patience}\")\n            if self.counter >= self.patience:\n                self.early_stop = True\n\n# # Training\n# model = FeedForwardBottleneck(input_dim=X_train.shape[1]).to(device)\n# optimizer = torch.optim.Adam(model.parameters(), lr=1e-4, weight_decay=1e-5)\n# epochs = 500\n# early_stopper = EarlyStopping(patience=10)\n\n# for epoch in range(epochs):\n#     model.train()\n#     total_loss = 0\n\n#     for xb, yb in train_loader:\n#         xb, yb = xb.to(device), yb.to(device)\n#         optimizer.zero_grad()\n#         preds, _ = model(xb)\n#         mse = F.mse_loss(preds, yb)\n#         p_loss = pearson_loss(preds, yb)\n#         loss = 0.7 * p_loss + 0.3 * mse\n#         loss.backward()\n#         optimizer.step()\n#         total_loss += loss.item()\n\n#     avg_train_loss = total_loss / len(train_loader)\n\n#     # val\n#     model.eval()\n#     val_loss = 0\n#     with torch.no_grad():\n#         for xb, yb in val_loader:\n#             xb, yb = xb.to(device), yb.to(device)\n#             preds, _ = model(xb)\n#             mse = F.mse_loss(preds, yb)\n#             p_loss = pearson_loss(preds, yb)\n#             loss = 0.7 * p_loss + 0.3 * mse\n#             val_loss += loss.item()\n#     avg_val_loss = val_loss / len(val_loader)\n\n#     print(f\"Epoch {epoch+1}/{epochs} | Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_val_loss:.4f}\")\n\n#     if early_stopper.best_loss is None or avg_val_loss < early_stopper.best_loss:\n#         torch.save(model.state_dict(), \"best_ff_model.pth\")\n#         print(\"✅ Model improved. Saved to best_ff_model.pth\")\n\n#     early_stopper(avg_val_loss)\n#     if early_stopper.early_stop:\n#         print(\"⏹️ Early stopping triggered. Training stopped.\")\n#         break\n\n# print(\"✅ Training finished.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:15.329586Z","iopub.execute_input":"2025-07-24T19:11:15.329838Z","iopub.status.idle":"2025-07-24T19:11:15.699272Z","shell.execute_reply.started":"2025-07-24T19:11:15.329815Z","shell.execute_reply":"2025-07-24T19:11:15.698720Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## NN Features & NN Predictions","metadata":{}},{"cell_type":"code","source":"model = FeedForwardBottleneck(input_dim=X_train.shape[1]).to(device)\nmodel.load_state_dict(torch.load(\"/kaggle/input/nn-model/best_ff_model.pth\", map_location=device))\nmodel.eval()\nprint(\"✅ Model loaded from best_ff_model.pth\")\n\n\n# result\n@torch.no_grad()\ndef extract_bottleneck_features(model, data_tensor, batch_size=512):\n    model.eval()\n    preds_list = []\n    z_list = []\n    for i in range(0, len(data_tensor), batch_size):\n        batch = data_tensor[i:i+batch_size].to(device)\n        preds, z = model(batch)\n        preds_list.append(preds.cpu())\n        z_list.append(z.cpu())\n    preds_all = torch.cat(preds_list).numpy()\n    z_all = torch.cat(z_list).numpy()\n    return preds_all, z_all\n\ntrain_tensor = torch.from_numpy(train_df[nn_input].values.astype(np.float32))\ntest_tensor = torch.from_numpy(test_df[nn_input].values.astype(np.float32))\n\nNN_train_preds, train_feats_np = extract_bottleneck_features(model, train_tensor)\nNN_test_preds, test_feats_np = extract_bottleneck_features(model, test_tensor)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:15.700243Z","iopub.execute_input":"2025-07-24T19:11:15.700423Z","iopub.status.idle":"2025-07-24T19:11:17.510986Z","shell.execute_reply.started":"2025-07-24T19:11:15.700409Z","shell.execute_reply":"2025-07-24T19:11:17.510349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_label_nn = test_label.copy()\ntest_label_nn[\"label\"] = NN_test_preds\nprint(test_label_nn.head(5))\ntest_label_nn = test_label_nn.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_label_nn.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_nn.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:17.515108Z","iopub.execute_input":"2025-07-24T19:11:17.515306Z","iopub.status.idle":"2025-07-24T19:11:18.461190Z","shell.execute_reply.started":"2025-07-24T19:11:17.515290Z","shell.execute_reply":"2025-07-24T19:11:18.460587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a new feature column for bottleneck features (N0, N1, ...)\nnew_feat_cols = [f\"N{i}\" for i in range(train_feats_np.shape[1])]\n\n# Directly add the new bottleneck features to the original DataFrames\nfor i, col in enumerate(new_feat_cols):\n    train_df[col] = train_feats_np[:, i]  # Add bottleneck features to train_df\n    test_df[col] = test_feats_np[:, i]    # Add bottleneck features to test_df\n\n# Print the final shapes to check if the features are added correctly\nprint(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:18.461875Z","iopub.execute_input":"2025-07-24T19:11:18.462060Z","iopub.status.idle":"2025-07-24T19:11:18.560364Z","shell.execute_reply.started":"2025-07-24T19:11:18.462045Z","shell.execute_reply":"2025-07-24T19:11:18.559630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cols = [f'N{i}' for i in range(16)]\n# zero_ratio = (train_df[cols] == 0).sum() / len(train_df)\n# print(\"Train zero_ratio in NN features:\", zero_ratio)\n\n# cols = [f'N{i}' for i in range(16)]\n# zero_ratio = (test_df[cols] == 0).sum() / len(test_df)\n# print(\"Test zero_ratio in NN features:\", zero_ratio)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:18.561138Z","iopub.execute_input":"2025-07-24T19:11:18.561330Z","iopub.status.idle":"2025-07-24T19:11:18.564904Z","shell.execute_reply.started":"2025-07-24T19:11:18.561313Z","shell.execute_reply":"2025-07-24T19:11:18.564196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_trans_Anony = [col for col in train_df.columns if col.startswith(\"N\")]\ncorrelations = train_df[new_trans_Anony].corrwith(train_df['label'])\ntop_5_features = correlations.abs().sort_values(ascending=False).head(5)\n\nprint(\"Top 5 features with the highest absolute correlation with label:\")\nfor feature, corr in top_5_features.items():\n    print(f\"{feature}: {corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:18.565631Z","iopub.execute_input":"2025-07-24T19:11:18.565940Z","iopub.status.idle":"2025-07-24T19:11:18.722211Z","shell.execute_reply.started":"2025-07-24T19:11:18.565914Z","shell.execute_reply":"2025-07-24T19:11:18.721219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_list = [f\"N{i}\" for i in range(16)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:18.723314Z","iopub.execute_input":"2025-07-24T19:11:18.723653Z","iopub.status.idle":"2025-07-24T19:11:18.727644Z","shell.execute_reply.started":"2025-07-24T19:11:18.723624Z","shell.execute_reply":"2025-07-24T19:11:18.727043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = base_cols + top_features + pc_list + n_list + ['label']\ntest_cols = base_cols + top_features + pc_list + n_list\n\ntrain_cols = [col for col in train_cols if col in train_df.columns]\ntest_cols = [col for col in test_cols if col in test_df.columns]\n\ntrain_df = train_df[train_cols].copy()\ntest_df = test_df[test_cols].copy()\n\nprint(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:18.728313Z","iopub.execute_input":"2025-07-24T19:11:18.728539Z","iopub.status.idle":"2025-07-24T19:11:19.010812Z","shell.execute_reply.started":"2025-07-24T19:11:18.728522Z","shell.execute_reply":"2025-07-24T19:11:19.009887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:19.011733Z","iopub.execute_input":"2025-07-24T19:11:19.012502Z","iopub.status.idle":"2025-07-24T19:11:19.178008Z","shell.execute_reply.started":"2025-07-24T19:11:19.012472Z","shell.execute_reply":"2025-07-24T19:11:19.177198Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Lag Features**\n## Specify Generation","metadata":{}},{"cell_type":"code","source":"def add_required_lag_features(train, test, train_df, test_df):\n    \n    # Original + new list of required features\n    features_to_create = [\n        'X198_lead_30', 'X179_lead_120', 'X197_lead_30',\n        'X173_lead_120', 'X179_lead_150', 'X198', 'X40',\n        'X445_lead_30', 'X175_lead_365', 'X181_lead_365',\n        'X240_lead_30', 'X239_lead_40', 'X119_lead_30', 'X239_lead_80',\n        'X113_lead_70', 'X239_lead_50', 'X623_lead_30', 'X623_lead_40',\n        'X119_lead_80', 'X157_lead_30', 'X119_lead_50', 'X173_lead_80',\n        'X215_lead_40', 'X239_lead_70', 'X623_lead_80', 'X743_lead_30',\n        'X239_lead_60', 'X173_lead_40', 'X324_lead_30', 'X240_lead_60',\n        'X113_lead_50', 'X743_lead_60', 'X743_lead_70', 'X198_lead_70',\n        'X743_lead_50', 'X240_lead_70', 'X198_lead_50', 'X173_lead_60',\n        'X173_lead_70', 'X157_lead_40', 'X198_lead_80', 'X198_lead_40',\n        'X743_lead_80', 'X240_lead_50', 'X113_lead_30', 'X119_lead_70',\n        'X240_lead_80', 'X240_lead_40', 'X239_lead_30'\n    ]\n\n    # Step 1: Separate lagged and direct (non-lagged) features\n    lag_map = {}        # Format: {lag: [column1, column2, ...]}\n    direct_cols = []    # Columns without lag\n\n    for feat in features_to_create:\n        if \"_lead_\" in feat:\n            col, lag = feat.split(\"_lead_\")\n            lag = int(lag)\n            lag_map.setdefault(lag, []).append(col)\n        else:\n            direct_cols.append(feat)\n\n    # Step 2: Combine train and test for consistent lag computation\n    full_df = pd.concat([train, test], axis=0, ignore_index=True)\n\n    lagged_features = []\n\n    # Step 3: Generate lagged features\n    for lag, cols in lag_map.items():\n        print(f\"📦 Creating lagged features: lag={lag}, columns={cols}\")\n        lagged = full_df[cols].shift(-lag)\n        lagged.columns = [f\"{col}_lead_{lag}\" for col in cols]\n        lagged = lagged.fillna(0.0).astype(np.float32)\n        lagged_features.append(lagged)\n\n    # Step 4: Combine all lagged features\n    lagged_df = pd.concat(lagged_features, axis=1)\n\n    # Step 5: Split lagged features back to train/test\n    train_lag = lagged_df.iloc[:len(train)].reset_index(drop=True)\n    test_lag = lagged_df.iloc[len(train):].reset_index(drop=True)\n\n    # Step 6: Extract direct features\n    direct_train = train[direct_cols].reset_index(drop=True)\n    direct_test = test[direct_cols].reset_index(drop=True)\n\n    # Step 7: Concatenate everything\n    train_df_final = pd.concat([train_df.reset_index(drop=True), direct_train, train_lag], axis=1)\n    test_df_final = pd.concat([test_df.reset_index(drop=True), direct_test, test_lag], axis=1)\n\n    # Step 8: Memory cleanup\n    gc.collect()\n\n    print(\"✅ Feature construction completed.\")\n    print(f\"train_df_final.shape = {train_df_final.shape}\")\n    print(f\"test_df_final.shape = {test_df_final.shape}\")\n\n    return train_df_final, test_df_final\n\ntrain_df, test_df = add_required_lag_features(train, test, train_df, test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:19.178957Z","iopub.execute_input":"2025-07-24T19:11:19.179251Z","iopub.status.idle":"2025-07-24T19:11:21.154101Z","shell.execute_reply.started":"2025-07-24T19:11:19.179226Z","shell.execute_reply":"2025-07-24T19:11:21.153103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Batch Generation","metadata":{}},{"cell_type":"code","source":"def append_lag_features(train, test, train_df, test_df, mi_list, lag_list, chunk_size=10):\n\n    full_df = pd.concat([train[mi_list], test[mi_list]], axis=0, ignore_index=True)\n    lag_features_all = []\n\n    total_chunks = len(range(0, len(lag_list), chunk_size))\n    print(f\"total {total_chunks} chunks\")\n\n    for i in tqdm(range(0, len(lag_list), chunk_size), desc=\"Lag chunck process\"):\n        lag_chunk = lag_list[i:i+chunk_size]\n        print(f\"  → creating lag feature: {lag_chunk}\")\n        \n        for lag in tqdm(lag_chunk, leave=False, desc=\"  sub process\"):\n            lagged = full_df.shift(-lag)\n            lagged.columns = [f\"{col}_lead_{lag}\" for col in lagged.columns]\n            lagged = lagged.fillna(0.0).astype(np.float32)\n            lag_features_all.append(lagged)\n\n        gc.collect()\n\n    all_lag_features = pd.concat(lag_features_all, axis=1)\n\n    train_lagged = all_lag_features.iloc[:len(train)].reset_index(drop=True)\n    test_lagged = all_lag_features.iloc[len(train):].reset_index(drop=True)\n\n    train_df_aug = pd.concat([train_df.reset_index(drop=True), train_lagged], axis=1)\n    test_df_aug = pd.concat([test_df.reset_index(drop=True), test_lagged], axis=1)\n\n    return train_df_aug, test_df_aug\n\n\n# train = train[mi_list]\n# test = test[mi_list]\n\n# lag_list = [1, 5, 15, 20, 30, 60, 120, 150, 300]\n# chunk_size = 3\n\n# train_df_lagged, test_df_lagged = append_lag_features(\n#     train, test, train_df, test_df, mi_list, lag_list, chunk_size\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.155226Z","iopub.execute_input":"2025-07-24T19:11:21.155959Z","iopub.status.idle":"2025-07-24T19:11:21.162940Z","shell.execute_reply.started":"2025-07-24T19:11:21.155930Z","shell.execute_reply":"2025-07-24T19:11:21.162116Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **LGBM: Time Series Splitting**","metadata":{}},{"cell_type":"code","source":"# train_cols = base_cols + top_features + pc_list + n_list + ['label']\n# test_cols = base_cols + top_features + pc_list + n_list\n\n# train_cols = [col for col in train_cols if col in train_df.columns]\n# test_cols = [col for col in test_cols if col in test_df.columns]\n\n# train_df_all = train_df[train_cols].copy()\n# test_df_all = test_df[test_cols].copy()\n\n# print(train_df_all.shape)\n# print(test_df_all.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.180166Z","iopub.status.idle":"2025-07-24T19:11:21.180433Z","shell.execute_reply.started":"2025-07-24T19:11:21.180305Z","shell.execute_reply":"2025-07-24T19:11:21.180319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.drop(columns=['label'])\ny = train_df['label']\n\nsplit_index = int(len(X) * 0.8)\nX_train, X_val = X.iloc[:split_index], X.iloc[split_index:]\ny_train, y_val = y.iloc[:split_index], y.iloc[split_index:]\n\ndef pearson_metric(y_true, y_pred):\n    score, _ = pearsonr(y_true, y_pred)\n    return 'pearson', score, True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.181663Z","iopub.status.idle":"2025-07-24T19:11:21.181995Z","shell.execute_reply.started":"2025-07-24T19:11:21.181844Z","shell.execute_reply":"2025-07-24T19:11:21.181857Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Optuna","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['label'])\ny = train_df['label']\n\nsplit_index = int(len(X) * 0.8)\nX_train, X_val = X.iloc[:split_index], X.iloc[split_index:]\ny_train, y_val = y.iloc[:split_index], y.iloc[split_index:]\n\ndef pearson_metric(y_true, y_pred):\n    score, _ = pearsonr(y_true, y_pred)\n    return 'pearson', score, True\n\ndef objective(trial):\n    params = {\n        'objective': 'regression',\n        'metric': 'None',  \n        'verbosity': -1,\n        'boosting_type': 'gbdt',\n        'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.3, log=True),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 300),\n        'max_depth': trial.suggest_int('max_depth', 3, 15),\n        'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 10.0, log=True),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 10.0, log=True),\n        'n_estimators': 1000,\n        'random_state': 42,\n        'n_jobs': 1,\n        'device': 'gpu',\n    }\n\n    try:\n        model = lgb.LGBMRegressor(**params)\n        model.fit(\n            X_train, y_train,\n            eval_set=[(X_val, y_val)],\n            eval_metric=pearson_metric,\n            callbacks=[lgb.early_stopping(50, verbose=False)]\n            \n        )\n\n        preds = model.predict(X_val)\n        score, _ = pearsonr(y_val, preds)\n\n        print(f\"Trial {trial.number}, Pearson score: {score:.4f}\")\n\n        del model, preds\n        gc.collect()\n\n        return float(score)\n\n    except Exception as e:\n        print(f\"Trial {trial.number} failed with exception: {e}\")\n        gc.collect()\n        return -1.0\n\n# study = optuna.create_study(direction='maximize')\n# study.optimize(objective, n_trials=20)\n\n# print(\"Best parameters:\", study.best_params)\n\n# best_params = study.best_params\n\n# best_params.update({\n#     'objective': 'regression',\n#     'metric': 'None',\n#     'boosting_type': 'gbdt',\n#     'n_estimators': 1000,\n#     'n_jobs': 1,\n#     'device': 'gpu',\n#     'verbosity': -1\n# })\n\nbest_params = {\n    'objective': 'regression',\n    'metric': 'None',\n    'boosting_type': 'gbdt',\n    'n_estimators': 1000,\n    'n_jobs': 1,\n    'device': 'gpu',\n    'verbosity': -1,\n    'learning_rate': 0.015865508582763952,\n    'num_leaves': 67,\n    'max_depth': 10,\n    'min_child_samples': 66,\n    'subsample': 0.6655741689043483,\n    'colsample_bytree': 0.5974224259462245,\n    'reg_alpha': 9.090859081454806e-06,\n    'reg_lambda': 2.3603142329797598e-07\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.183312Z","iopub.status.idle":"2025-07-24T19:11:21.184014Z","shell.execute_reply.started":"2025-07-24T19:11:21.183825Z","shell.execute_reply":"2025-07-24T19:11:21.183843Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5-fold Expansion Window","metadata":{}},{"cell_type":"code","source":"def custom_time_series_split(X, folds=5, val_ratio=0.2):\n    n = len(X)\n    fold_sizes = [n // folds] * folds\n    for i in range(n % folds):\n        fold_sizes[i] += 1\n    \n    fold_boundaries = []\n    start = 0\n    for size in fold_sizes:\n        fold_boundaries.append((start, start + size))\n        start += size\n\n    splits = []\n    for (start, end) in fold_boundaries:\n        fold_len = end - start\n        val_start = start + int(fold_len * (1 - val_ratio))\n\n        train_idx = np.arange(0, val_start)\n        val_idx = np.arange(val_start, end)\n\n        splits.append((train_idx, val_idx))\n\n    return splits\n\n\ndef cross_val_lgb_time_series_weighted(X_tr, y_tr, X_test, params, folds=5, val_ratio=0.2):\n    splits = custom_time_series_split(X_tr, folds=folds, val_ratio=val_ratio)\n\n    cv_scores, fold_preds = [], []\n    feat_imp = np.zeros(X_tr.shape[1])\n    feature_names = X_tr.columns.tolist()\n\n    top30_nonanon_per_fold = []\n\n    X_test = X_test.reindex(columns=X_tr.columns, fill_value=0.0)\n\n    for fold, (tr_idx, va_idx) in enumerate(splits):\n        X_tr_fold, X_va_fold = X_tr.iloc[tr_idx], X_tr.iloc[va_idx]\n        y_tr_fold, y_va_fold = y_tr.iloc[tr_idx], y_tr.iloc[va_idx]\n\n        print(f\"[Fold {fold+1}] Train idx: {tr_idx[0]}-{tr_idx[-1]}, Val idx: {va_idx[0]}-{va_idx[-1]}\")\n\n        model = lgb.LGBMRegressor(**params, random_state=42 + fold)\n        model.fit(\n            X_tr_fold, y_tr_fold,\n            eval_set=[(X_va_fold, y_va_fold)],\n            eval_metric=pearson_metric,\n            callbacks=[lgb.early_stopping(50, verbose=False)]\n        )\n\n        fold_feat_imp = model.booster_.feature_importance(importance_type=\"gain\")\n        feat_imp += fold_feat_imp\n\n        imp_df = pd.DataFrame({\n            'feature': feature_names,\n            'importance': fold_feat_imp\n        }).sort_values('importance', ascending=False)\n\n        # ✅ not \"PC\"\n        non_anon_imp_df = imp_df[~imp_df['feature'].str.startswith(\"Q\")].copy()\n        \n        print(f\"[Fold {fold+1}] Top 10 non-anonymous features:\")\n        print(non_anon_imp_df.head(10).to_string(index=False))\n\n        top30 = non_anon_imp_df.head(30).reset_index(drop=True)\n        top30_nonanon_per_fold.append(top30)\n\n        val_pred = model.predict(X_va_fold)\n        score = pearsonr(y_va_fold, val_pred)[0]\n        cv_scores.append(score)\n        print(f\"[Fold {fold+1}] Pearson Score: {score:.4f}\")\n\n        fold_preds.append(model.predict(X_test))\n\n        del model\n        gc.collect()\n\n    # ---------- Weighted embedding ----------\n    weights = np.array([0, 0.4, 0, 0.4, 0.2], dtype=float)\n    weights = weights[:len(fold_preds)]\n    weights /= weights.sum()\n    preds_test = np.average(fold_preds, axis=0, weights=weights)\n\n    mean_cv = np.mean(cv_scores)\n    avg_feat_imp = feat_imp / len(fold_preds)\n\n    # ---------- The intersection of the last two folds is not an anonymous feature ----------\n    if len(top30_nonanon_per_fold) >= 2:\n        last_two = top30_nonanon_per_fold[-2:]\n        set1 = set(last_two[0]['feature'])\n        set2 = set(last_two[1]['feature'])\n        common_features = list(set1.intersection(set2))\n\n        avg_importance_dict = {}\n        for feat in common_features:\n            imp1 = last_two[0].loc[last_two[0]['feature'] == feat, 'importance'].values[0]\n            imp2 = last_two[1].loc[last_two[1]['feature'] == feat, 'importance'].values[0]\n            avg_importance_dict[feat] = (imp1 + imp2) / 2\n\n        common_imp_df = pd.DataFrame({\n            'feature': list(avg_importance_dict.keys()),\n            'avg_importance': list(avg_importance_dict.values())\n        }).sort_values('avg_importance', ascending=False)\n\n        print(\"\\n✅ Top common non-anonymous features in last two folds (intersection):\")\n        print(common_imp_df.to_string(index=False))\n\n    return mean_cv, preds_test, avg_feat_imp\n\nmean_cv, LGB_NN_5f_preds, imp_gain = cross_val_lgb_time_series_weighted(\n    X, y, test_df, best_params, folds=5, val_ratio=0.2\n)\n\nprint(f\"Mean CV Pearson: {mean_cv:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.185106Z","iopub.status.idle":"2025-07-24T19:11:21.185400Z","shell.execute_reply.started":"2025-07-24T19:11:21.185258Z","shell.execute_reply":"2025-07-24T19:11:21.185273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_label_lgb_nn_5f = test_label.copy()\ntest_label_lgb_nn_5f[\"label\"] = LGB_NN_5f_preds\nprint(test_label_lgb_nn_5f.head(5))\ntest_label_lgb_nn_5f = test_label_lgb_nn_5f.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_label_lgb_nn_5f.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_lgb_nn_5f.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.186234Z","iopub.status.idle":"2025-07-24T19:11:21.186437Z","shell.execute_reply.started":"2025-07-24T19:11:21.186342Z","shell.execute_reply":"2025-07-24T19:11:21.186350Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-fold Rolling Window","metadata":{}},{"cell_type":"code","source":"def custom_time_series_split(X, folds=3, val_ratio=0.2):\n    \"\"\"\n    Sliding window time series split:\n    - Each fold is a fixed-size chunk of data (e.g. 1/3 of total)\n    - In each fold, use the first (1 - val_ratio) as training and the rest as validation\n    \"\"\"\n    import numpy as np\n\n    n = len(X)\n    fold_size = n // folds\n\n    splits = []\n    for i in range(folds):\n        fold_start = i * fold_size\n        fold_end = (i + 1) * fold_size if i < folds - 1 else n\n\n        fold_indices = np.arange(fold_start, fold_end)\n        fold_len = fold_end - fold_start\n\n        val_len = int(fold_len * val_ratio)\n        train_len = fold_len - val_len\n\n        train_idx = fold_indices[:train_len]\n        val_idx = fold_indices[train_len:]\n\n        splits.append((train_idx, val_idx))\n\n    return splits\n\n\ndef cross_val_lgb_time_series_weighted(X_tr, y_tr, X_test, params, folds=3):\n    \"\"\"\n    Perform 3-fold sliding window cross-validation with weighted ensemble.\n    \"\"\"\n    splits = custom_time_series_split(X_tr, folds=folds)\n\n    cv_scores, fold_preds = [], []\n    feat_imp = np.zeros(X_tr.shape[1])\n    feature_names = X_tr.columns.tolist()\n\n    top30_nonanon_per_fold = []\n\n    # Align test set columns\n    X_test = X_test.reindex(columns=X_tr.columns, fill_value=0.0)\n\n    for fold, (tr_idx, va_idx) in enumerate(splits):\n        X_tr_fold, X_va_fold = X_tr.iloc[tr_idx], X_tr.iloc[va_idx]\n        y_tr_fold, y_va_fold = y_tr.iloc[tr_idx], y_tr.iloc[va_idx]\n\n        print(f\"[Fold {fold+1}] Train idx: {tr_idx[0]}-{tr_idx[-1]}, Val idx: {va_idx[0]}-{va_idx[-1]}\")\n\n        model = lgb.LGBMRegressor(**params, random_state=42 + fold)\n        model.fit(\n            X_tr_fold, y_tr_fold,\n            eval_set=[(X_va_fold, y_va_fold)],\n            eval_metric=pearson_metric,\n            callbacks=[lgb.early_stopping(50, verbose=False)]\n        )\n\n        fold_feat_imp = model.booster_.feature_importance(importance_type=\"gain\")\n        feat_imp += fold_feat_imp\n\n        imp_df = pd.DataFrame({\n            'feature': feature_names,\n            'importance': fold_feat_imp\n        }).sort_values('importance', ascending=False)\n\n        # Filter non-anonymous features\n        non_anon_imp_df = imp_df[~imp_df['feature'].str.startswith(\"Q\")].copy()\n        \n        print(f\"[Fold {fold+1}] Top 10 non-anonymous features:\")\n        print(non_anon_imp_df.head(10).to_string(index=False))\n\n        top30 = non_anon_imp_df.head(30).reset_index(drop=True)\n        top30_nonanon_per_fold.append(top30)\n\n        val_pred = model.predict(X_va_fold)\n        score = pearsonr(y_va_fold, val_pred)[0]\n        cv_scores.append(score)\n        print(f\"[Fold {fold+1}] Pearson Score: {score:.4f}\")\n\n        fold_preds.append(model.predict(X_test))\n\n        del model\n        gc.collect()\n\n    # ---------- Weighted ensemble ----------\n    # Adjust the weights for the last two folds (0, 0.5, 0.5)\n    weights = np.array([0, 0, 1], dtype=float)  # 3 folds with 0, 0.5, 0.5 weights\n    weights = weights[:len(fold_preds)]  # Ensure that we only use the folds we have\n    weights /= weights.sum()  # Normalize the weights\n    preds_test = np.average(fold_preds, axis=0, weights=weights)\n\n    mean_cv = np.mean(cv_scores)\n    avg_feat_imp = feat_imp / len(fold_preds)\n\n    # ---------- Intersection of the last two folds (non-anonymous features) ----------\n    if len(top30_nonanon_per_fold) >= 2:\n        last_two = top30_nonanon_per_fold[-2:]\n        set1 = set(last_two[0]['feature'])\n        set2 = set(last_two[1]['feature'])\n        common_features = list(set1.intersection(set2))\n\n        avg_importance_dict = {}\n        for feat in common_features:\n            imp1 = last_two[0].loc[last_two[0]['feature'] == feat, 'importance'].values[0]\n            imp2 = last_two[1].loc[last_two[1]['feature'] == feat, 'importance'].values[0]\n            avg_importance_dict[feat] = (imp1 + imp2) / 2\n\n        common_imp_df = pd.DataFrame({\n            'feature': list(avg_importance_dict.keys()),\n            'avg_importance': list(avg_importance_dict.values())\n        }).sort_values('avg_importance', ascending=False)\n\n        print(\"\\n✅ Top common non-anonymous features in last two folds (intersection):\")\n        print(common_imp_df.to_string(index=False))\n\n    return mean_cv, preds_test, avg_feat_imp\n\n# Example usage\nmean_cv, LGB_NN_3f_preds, imp_gain = cross_val_lgb_time_series_weighted(\n    X, y, test_df, best_params, folds=3\n)\n\nprint(f\"Mean CV Pearson: {mean_cv:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.188060Z","iopub.status.idle":"2025-07-24T19:11:21.188303Z","shell.execute_reply.started":"2025-07-24T19:11:21.188188Z","shell.execute_reply":"2025-07-24T19:11:21.188200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_label_lgb_nn_3f = test_label.copy()\ntest_label_lgb_nn_3f[\"label\"] = LGB_NN_3f_preds\nprint(test_label_lgb_nn_3f.head(5))\ntest_label_lgb_nn_3f = test_label_lgb_nn_3f.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_label_lgb_nn_3f.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_lgb_nn_3f.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.188938Z","iopub.status.idle":"2025-07-24T19:11:21.189143Z","shell.execute_reply.started":"2025-07-24T19:11:21.189042Z","shell.execute_reply":"2025-07-24T19:11:21.189051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_lgb_nn_combo = test_label.copy()\ncombo = (LGB_NN_3f_preds + LGB_NN_5f_preds)/2\ntest_lgb_nn_combo[\"label\"] = combo\nprint(test_lgb_nn_combo.head(5))\ntest_lgb_nn_combo = test_lgb_nn_combo.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_lgb_nn_combo.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_lgb_nn_combo.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:21.190734Z","iopub.status.idle":"2025-07-24T19:11:21.191121Z","shell.execute_reply.started":"2025-07-24T19:11:21.190945Z","shell.execute_reply":"2025-07-24T19:11:21.190961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Linear Model**\n## Ridege","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['label'])\ny = train_df['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:49.746287Z","iopub.execute_input":"2025-07-24T19:11:49.746575Z","iopub.status.idle":"2025-07-24T19:11:49.798333Z","shell.execute_reply.started":"2025-07-24T19:11:49.746552Z","shell.execute_reply":"2025-07-24T19:11:49.797822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\n\nX_scaled = scaler.fit_transform(X)\nX_scaled = pd.DataFrame(X_scaled, columns=X.columns, index=X.index)\n\ntest_scaled = scaler.transform(test_df)\ntest_scaled = pd.DataFrame(test_scaled, columns=test_df.columns, index=test_df.index)\n\nmodel = Ridge()\nmodel.fit(X_scaled, y)\n\ncoef_series = pd.Series(model.coef_, index=X.columns).sort_values()\n\nthreshold = 0.05\nlow_coef_features = coef_series[coef_series.abs() < threshold].index\n\nX_cleaned = X_scaled.drop(columns=low_coef_features)\ntest_cleaned = test_scaled.drop(columns=low_coef_features)\n\nmodel_cleaned = Ridge()\nmodel_cleaned.fit(X_cleaned, y)\n\ncoef_series_cleaned = pd.Series(model_cleaned.coef_, index=X_cleaned.columns).sort_values()\n\nplt.figure(figsize=(15, 4))\ncoef_series_cleaned.plot(kind='bar')\nplt.title(\"Feature Importance (Ridge Coefficients) After Filtering\")\nplt.xlabel(\"Feature\")\nplt.ylabel(\"Coefficient Value\")\nplt.tight_layout()\nplt.show()\n\nprint(f\"Removed features with abs(coef) < {threshold}:\\n{low_coef_features.tolist()}\")\n\nridge_preds = model_cleaned.predict(test_cleaned)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:11:53.101881Z","iopub.execute_input":"2025-07-24T19:11:53.102135Z","iopub.status.idle":"2025-07-24T19:11:55.408310Z","shell.execute_reply.started":"2025-07-24T19:11:53.102119Z","shell.execute_reply":"2025-07-24T19:11:55.407628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ridge = test_label.copy()\ntest_ridge[\"label\"] = ridge_preds\nprint(test_ridge.head(5))\ntest_ridge = test_ridge.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_ridge.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_ridge.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:12:10.298089Z","iopub.execute_input":"2025-07-24T19:12:10.298668Z","iopub.status.idle":"2025-07-24T19:12:11.532218Z","shell.execute_reply.started":"2025-07-24T19:12:10.298646Z","shell.execute_reply":"2025-07-24T19:12:11.531374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## SGDRegressor","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['label'])\ny = train_df['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:13:38.092593Z","iopub.execute_input":"2025-07-24T19:13:38.092923Z","iopub.status.idle":"2025-07-24T19:13:38.143827Z","shell.execute_reply.started":"2025-07-24T19:13:38.092902Z","shell.execute_reply":"2025-07-24T19:13:38.143066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\n\nX_scaled = scaler.fit_transform(X)\nX_scaled = pd.DataFrame(X_scaled, columns=X.columns, index=X.index)\n\ntest_scaled = scaler.transform(test_df)\ntest_scaled = pd.DataFrame(test_scaled, columns=test_df.columns, index=test_df.index)\n\nmodel = SGDRegressor(random_state=42)\nmodel.fit(X_scaled, y)\n\ncoef_series = pd.Series(model.coef_, index=X.columns).sort_values()\n\nthreshold = 0.05\nlow_coef_features = coef_series[coef_series.abs() < threshold].index\n\nX_cleaned = X_scaled.drop(columns=low_coef_features)\ntest_cleaned = test_scaled.drop(columns=low_coef_features)\n\nmodel_cleaned = SGDRegressor(random_state=42)\nmodel_cleaned.fit(X_cleaned, y)\n\ncoef_series_cleaned = pd.Series(model_cleaned.coef_, index=X_cleaned.columns).sort_values()\n\nplt.figure(figsize=(15, 4))\ncoef_series_cleaned.plot(kind='bar')\nplt.title(\"Feature Importance (Coefficients) After Filtering\")\nplt.xlabel(\"Feature\")\nplt.ylabel(\"Coefficient Value\")\nplt.tight_layout()\nplt.show()\n\nprint(f\"Removed features with abs(coef) < {threshold}:\\n{low_coef_features.tolist()}\")\n\nSGD_preds = model_cleaned.predict(test_cleaned)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:13:59.400146Z","iopub.execute_input":"2025-07-24T19:13:59.400886Z","iopub.status.idle":"2025-07-24T19:14:05.461433Z","shell.execute_reply.started":"2025-07-24T19:13:59.400858Z","shell.execute_reply":"2025-07-24T19:14:05.460626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_SGD = test_label.copy()\ntest_SGD[\"label\"] = SGD_preds\nprint(test_SGD.head(5))\ntest_SGD = test_SGD.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_SGD.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_SGD.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:14:09.340774Z","iopub.execute_input":"2025-07-24T19:14:09.341482Z","iopub.status.idle":"2025-07-24T19:14:10.589669Z","shell.execute_reply.started":"2025-07-24T19:14:09.341455Z","shell.execute_reply":"2025-07-24T19:14:10.588845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_combo = 0*ridge_preds + 1*SGD_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:16:10.472354Z","iopub.execute_input":"2025-07-24T19:16:10.472637Z","iopub.status.idle":"2025-07-24T19:16:10.477977Z","shell.execute_reply.started":"2025-07-24T19:16:10.472615Z","shell.execute_reply":"2025-07-24T19:16:10.477347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_linear_combo = test_label.copy()\ntest_linear_combo[\"label\"] = linear_combo\nprint(test_linear_combo.head(5))\ntest_linear_combo = test_linear_combo.reset_index()\nsubmit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmit['prediction'] = test_linear_combo.set_index('ID').loc[submit['ID'], 'label'].values\nsubmit = submit.sort_values(by='ID').reset_index(drop=True)\nprint(submit.head(5))\nsubmit.to_csv(\"submission_linear_combo.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:16:56.992248Z","iopub.execute_input":"2025-07-24T19:16:56.992515Z","iopub.status.idle":"2025-07-24T19:16:58.206362Z","shell.execute_reply.started":"2025-07-24T19:16:56.992498Z","shell.execute_reply":"2025-07-24T19:16:58.205555Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Results Analysis**\n## Results of my Work\n\n\n* <span style=\"color:red\">submission_lgb_nn_combo.csv --> **0.84523**</span>\n* <span style=\"color:red\">submission_linear_combo.csv --> **0.83689**</span>\n","metadata":{}},{"cell_type":"code","source":"def analyze_series(data, title_prefix=\"\"):\n    data = np.asarray(data)\n    cumsum = np.cumsum(data)\n\n    fig, axs = plt.subplots(1, 3, figsize=(15, 3))\n\n    axs[0].plot(data, color='blue')\n    axs[0].set_title(f'{title_prefix} Line Plot')\n    axs[0].set_xlabel('Index')\n    axs[0].set_ylabel('Value')\n\n    axs[1].hist(data, bins=100, color='orange', alpha=0.7)\n    axs[1].set_title(f'{title_prefix} Histogram')\n    axs[1].set_xlabel('Value')\n    axs[1].set_ylabel('Frequency')\n\n    axs[2].plot(cumsum, color='green')\n    axs[2].set_title(f'{title_prefix} Cumulative Sum Plot')\n    axs[2].set_xlabel('Index')\n    axs[2].set_ylabel('Cumulative Sum')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:19:46.439402Z","iopub.execute_input":"2025-07-24T19:19:46.440313Z","iopub.status.idle":"2025-07-24T19:19:46.446219Z","shell.execute_reply.started":"2025-07-24T19:19:46.440284Z","shell.execute_reply":"2025-07-24T19:19:46.445254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyze_series(NN_test_preds, title_prefix=\"NN Test Predictions\")\nanalyze_series(LGB_NN_5f_preds, title_prefix=\"LGB NN 5 folds Test Predictions\")\nanalyze_series(LGB_NN_3f_preds, title_prefix=\"LGB NN 3 folds Test Predictions\")\nanalyze_series(combo, title_prefix=\"LGB NN combo Test Predictions\")\nanalyze_series(ridge_preds, title_prefix=\"Ridge Test Predictions\")\nanalyze_series(SGD_preds, title_prefix=\"SGD Test Predictions\")\nanalyze_series(linear_combo, title_prefix=\"Linear combo Test Predictions\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:20:35.719944Z","iopub.execute_input":"2025-07-24T19:20:35.720228Z","iopub.status.idle":"2025-07-24T19:20:38.142690Z","shell.execute_reply.started":"2025-07-24T19:20:35.720206Z","shell.execute_reply":"2025-07-24T19:20:38.141755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Public Results\nThis is a very interesting phenomenon. One prediction result with a score of 0.9 was obtained from [https://www.kaggle.com/code/taylorsamarel/reboot-sgd-convergence-not-blend-solution/notebook?scriptVersionId=250754938](http://), and the other with a score of 0.95 was from [https://www.kaggle.com/code/nina2025/drw-blend-horizontal-blend-vertical](http://). The correlation between their predicted values (return) is very high, but the cumulative predicted values (cum_return) show an almost negative correlation.","metadata":{}},{"cell_type":"code","source":"last_pred1 = pd.read_csv(\"/kaggle/input/preds-0-95/submission (24).csv\")\ntest_label_last1 = test_label.copy()\nlast_pred1 = last_pred1.set_index('ID')\nlast_pred1.index = last_pred1.index.astype(test_label_last1.index.dtype)\ntest_label_last1['label'] = last_pred1['prediction']\ntest_label_last1 = test_label_last1.reset_index()\nprint(test_label_last1.head())\n\nlast_pred2 = pd.read_csv(\"/kaggle/input/preds-0-90/submission.csv\")\ntest_label_last2 = test_label.copy()\nlast_pred2 = last_pred2.set_index('ID')\nlast_pred2.index = last_pred2.index.astype(test_label_last2.index.dtype)\ntest_label_last2['label'] = last_pred2['prediction']\ntest_label_last2 = test_label_last2.reset_index()\nprint(test_label_last2.head())\n\nlast_pred3 = pd.read_csv(\"/kaggle/input/preds-0-9516/submission (25).csv\")\ntest_label_last3 = test_label.copy()\nlast_pred3 = last_pred3.set_index('ID')\nlast_pred3.index = last_pred3.index.astype(test_label_last3.index.dtype)\ntest_label_last3['label'] = last_pred3['prediction']\ntest_label_last3 = test_label_last3.reset_index()\nprint(test_label_last3.head())\n\nanalyze_series(test_label_last1['label'], title_prefix=\"0.95_preds\")\nanalyze_series(test_label_last2['label'], title_prefix=\"0.90_preds\")\nanalyze_series(test_label_last3['label'], title_prefix=\"0.9516_preds\")\n\ncorrelation = test_label_last1['label'].corr(test_label_last2['label'])\nprint(\"preds corr：\",correlation)\ncumsum1 = test_label_last1['label'].cumsum()\ncumsum2 = test_label_last2['label'].cumsum()\ncorrelation = cumsum1.corr(cumsum2)\nprint(\"cum_preds corr：\", correlation)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:26:03.081024Z","iopub.execute_input":"2025-07-24T19:26:03.081827Z","iopub.status.idle":"2025-07-24T19:26:05.641186Z","shell.execute_reply.started":"2025-07-24T19:26:03.081801Z","shell.execute_reply":"2025-07-24T19:26:05.640300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Embedding**","metadata":{}},{"cell_type":"code","source":"del NN_test_preds, LGB_NN_5f_preds, LGB_NN_3f_preds, combo, ridge_preds, SGD_preds, linear_combo","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_nn = pd.read_csv(\"/kaggle/working/submission_lgb_nn_combo.csv\")\nridge_sgd = pd.read_csv(\"/kaggle/working/submission_linear_combo.csv\")\npublic1 = pd.read_csv(\"/kaggle/input/preds-0-95/submission (24).csv\")\npublic2 = pd.read_csv(\"/kaggle/input/preds-0-90/submission.csv\")\npublic3 = pd.read_csv(\"/kaggle/input/preds-0-9516/submission (25).csv\")\npublic4 = pd.read_csv(\"/kaggle/input/preds-0-95169/submission (26).csv\")\n\nweights = {\n    'lgb_nn': 0.1,\n    'ridge_sgd': 0.1,\n    'public1': 0.1,\n    'public2': 0.1,\n    'public3': 0.2, \n    'public4': 0.4,}\n\n\nassert (lgb_nn['ID'] == ridge_sgd['ID']).all()\nassert (lgb_nn['ID'] == public1['ID']).all()\nassert (lgb_nn['ID'] == public2['ID']).all()\nassert (lgb_nn['ID'] == public3['ID']).all()\nassert (lgb_nn['ID'] == public4['ID']).all()\n\nfinal_prediction = (\n    weights['lgb_nn'] * lgb_nn['prediction'] +\n    weights['ridge_sgd'] * ridge_sgd['prediction'] +\n    weights['public1'] * public1['prediction'] +\n    weights['public2'] * public2['prediction'] +\n    weights['public3'] * public3['prediction'] +\n    weights['public4'] * public4['prediction']\n)\n\n\nfinal_submission = pd.DataFrame({\n    'ID': lgb_nn['ID'],\n    'prediction': final_prediction\n})\n\nprint(final_submission.head(5))\n\nfinal_submission.to_csv(\"/kaggle/working/final_weighted_submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T19:36:26.337763Z","iopub.execute_input":"2025-07-24T19:36:26.338389Z","iopub.status.idle":"2025-07-24T19:36:28.006516Z","shell.execute_reply.started":"2025-07-24T19:36:26.338363Z","shell.execute_reply":"2025-07-24T19:36:28.005946Z"}},"outputs":[],"execution_count":null}]}